From 2542532ff8136200093f42ed5f84dcb6267960ae Mon Sep 17 00:00:00 2001 From: Ben Reitz Date: Tue, 8 Sep 2026 14:38:08 +0100 Subject: [PATCH 1/3] =?UTF-8?q?feat:=20durable=20REPL=20js=20tool=20?= =?UTF-8?q?=E2=80=94=20help(),=20emit(),=20framework-agnostic=20definition?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The model-facing tool is a framework-agnostic definition (name, description, plain JSON Schema, run) with zero framework dependencies; the AI SDK tool is a few-line shim over it, and pi/MCP wrappers are the same pattern. The description is generated from the grant set. In-session built-ins, neither recorded: help() renders grants, methods, docs, and data snapshots from the cell's injected shapes (replayed cells see the snapshot they were recorded under, so committed help() calls survive revocation and restart); emit(value) publishes ordered structured results alongside the cell value — replayed cells' emits are discarded like their console output, oversized or unclonable values degrade to their rendering, and past the per-cell cap emit throws instead of dropping. renderJsResultText() gives every text-only shim one shared rendering. --- packages/computer/src/index.ts | 9 + packages/computer/src/repl/runner.ts | 132 +++++++++- packages/computer/src/repl/session.ts | 4 +- packages/computer/src/repl/tool.test.ts | 169 +++++++++++++ packages/computer/src/repl/tool.ts | 196 +++++++++++++++ packages/computer/src/repl/types.ts | 15 +- packages/computer/src/tools/index.ts | 1 + packages/computer/src/tools/js.test.ts | 54 ++++ packages/computer/src/tools/js.ts | 26 ++ packages/computer/tests/repl-builtins.test.ts | 231 ++++++++++++++++++ packages/computer/tests/repl-worker.ts | 31 ++- packages/computer/vitest.config.repl.ts | 2 +- 12 files changed, 864 insertions(+), 6 deletions(-) create mode 100644 packages/computer/src/repl/tool.test.ts create mode 100644 packages/computer/src/repl/tool.ts create mode 100644 packages/computer/src/tools/js.test.ts create mode 100644 packages/computer/src/tools/js.ts create mode 100644 packages/computer/tests/repl-builtins.test.ts diff --git a/packages/computer/src/index.ts b/packages/computer/src/index.ts index d6db10ee..f2cf0c11 100644 --- a/packages/computer/src/index.ts +++ b/packages/computer/src/index.ts @@ -74,6 +74,15 @@ export { type ReplEvalOptions, type ReplSessionOptions, } from "./repl/session.js"; +export { + createJsToolDefinition, + renderJsResultText, + type JsToolDefinition, + type JsToolInput, + type JsToolOptions, + type JsToolSessionLike, + type JsToolWorkspaceLike, +} from "./repl/tool.js"; export type { ReplEffect, ReplErrorKind, diff --git a/packages/computer/src/repl/runner.ts b/packages/computer/src/repl/runner.ts index d2b6b492..c3a2bc7e 100644 --- a/packages/computer/src/repl/runner.ts +++ b/packages/computer/src/repl/runner.ts @@ -55,6 +55,12 @@ export function replCellsModule(count: number): string { const MAX_LOG_ENTRIES = 1_000; const MAX_LOG_ENTRY_CHARS = 8_192; +// emit() is load-bearing output (unlike logs), so past the per-cell entry +// cap it throws instead of dropping — deterministic across replay, and the +// error names the fix. The value ceiling is display transport, not replay +// input, so oversized values degrade to their rendering (omit-plus-render). +const MAX_EMIT_ENTRIES = 1_000; +const MAX_EMIT_VALUE_BYTES = 262_144; // The runner source. A template-built string (matching how the // worker-javascript backend ships its runtime module) so the package build @@ -119,6 +125,8 @@ function effect(kind, make) { let BRIDGE; const PROXY_META = new WeakMap(); const INJECTED = new Map(); +// Grant shapes injected for the cell currently running — help() reads them. +let CURRENT_SHAPES = null; function joinPath(path, prop) { return path === "" ? prop : path + "." + prop; @@ -207,6 +215,7 @@ async function doCall(id, path, args, recipe) { // cell start in both modes, so grants win over leftover same-name bindings // identically live and on replay. function injectGrants(shapes) { + CURRENT_SHAPES = shapes || null; // Restore whatever a grant name shadowed (e.g. the ambient-fetch error // when a capability was granted as \`fetch\`), then remove our proxies. for (const [name, prior] of INJECTED) { @@ -330,6 +339,120 @@ function inspect(value, depth) { return name + "{ " + entries.join(", ") + " }"; } +// --- in-session built-ins: help() and emit() ------------------------------ +// +// Neither is a capability and neither is ever recorded. help() is a pure +// function of the cell's injected grant shapes — replayed cells see the +// snapshot they were recorded under, so a committed help() call reproduces +// exactly even after grants change. emit() fills a per-cell display buffer; +// replayed cells' emits are discarded like their console output, so only +// the new cell's emits ride back on the result. + +const emits = { entries: [] }; + +function emitValue(value) { + if (emits.entries.length >= ${MAX_EMIT_ENTRIES}) { + throw new Error( + "emit(): this cell already emitted ${MAX_EMIT_ENTRIES} results. " + + "Emit fewer, larger entries — or write bulk data to a workspace file " + + "via a granted capability and emit the path." + ); + } + let text = inspect(value, 0); + if (text.length > ${MAX_LOG_ENTRY_CHARS}) text = text.slice(0, ${MAX_LOG_ENTRY_CHARS}) + "…"; + const entry = { text }; + attachEmitValue(entry, value); + emits.entries.push(entry); + return value; +} + +// Attach the structured value when it can cross the boundary and fits the +// per-entry ceiling; otherwise the rendering alone ships. Cloned at emit +// time, so later cell code can't mutate what was already published. +function attachEmitValue(entry, value) { + if (PROXY_META.has(value)) return; // capability handles are not values + let clone; + try { + clone = structuredClone(value); + } catch { + return; // unclonable: rendering only + } + try { + const estimate = JSON.stringify(value); + if (estimate !== undefined && estimate.length > ${MAX_EMIT_VALUE_BYTES}) { + entry.text += " (structured value omitted: about " + estimate.length + + " bytes exceeds the ${MAX_EMIT_VALUE_BYTES}-byte emit ceiling — " + + "write large data to a workspace file and emit the path)"; + return; + } + } catch { + // Clonable but not JSON-estimable (bigint, cycles): ship it. + } + entry.value = clone; +} + +function helpText(name) { + const shapes = CURRENT_SHAPES || {}; + const names = Object.keys(shapes); + if (name === undefined) { + const lines = [ + "Persistent JavaScript session: top-level await works, and bindings survive across cells — including restarts.", + "Built-ins:", + " help(\\"name\\") — full docs for one granted capability", + " emit(value) — publish an extra structured result alongside the cell value (returns the value)", + ]; + if (names.length === 0) { + lines.push("No capabilities are granted in this session — pure JavaScript with durable state."); + } else { + lines.push("Capabilities:"); + for (const n of names) { + const d = shapes[n] && shapes[n].description; + lines.push(" " + n + (d ? " — " + d : "")); + } + lines.push("help(\\"name\\") shows a capability's methods and docs."); + } + return lines.join("\\n"); + } + const key = String(name); + if (!Object.prototype.hasOwnProperty.call(shapes, key)) { + return "No capability named " + JSON.stringify(key) + " in this session." + + (names.length > 0 ? " Granted: " + names.join(", ") + "." : " No capabilities are granted.") + + " help() lists everything."; + } + const shape = shapes[key]; + const lines = [key + (shape.description ? " — " + shape.description : "")]; + renderShapeHelp(lines, key, "", shape, shape.docs || {}); + return lines.join("\\n"); +} + +// Walk a grant shape: methods (with grantor docs, dotted keys for nested +// surfaces), data snapshots with their current values, and children. +function renderShapeHelp(lines, path, docPath, shape, docs) { + if (shape.opaque) { + lines.push(" " + path + ".(…) — surface unknown (opaque remote stub): call any method it supports"); + return; + } + if (shape.callable) { + lines.push(" " + path + "(…) — callable directly" + (docs[docPath] ? ": " + docs[docPath] : "")); + } + for (const m of shape.methods || []) { + const dk = docPath === "" ? m : docPath + "." + m; + lines.push(" " + path + "." + m + "(…)" + (docs[dk] ? " — " + docs[dk] : "")); + } + for (const k of Object.keys(shape.data || {})) { + let rendered; + try { rendered = inspect(decodeReplValue(shape.data[k]), 1); } catch { rendered = "…"; } + if (rendered.length > 200) rendered = rendered.slice(0, 200) + "…"; + lines.push(" " + path + "." + k + " = " + rendered); + } + for (const c of Object.keys(shape.children || {})) { + renderShapeHelp(lines, path + "." + c, docPath === "" ? c : docPath + "." + c, shape.children[c], docs); + } +} + +globalThis.help = helpText; +globalThis.emit = emitValue; + function describeError(error, kind) { const isError = error instanceof Error; return { @@ -354,6 +477,7 @@ export default class ReplRunner extends WorkerEntrypoint { for (let i = 0; i < cells.length - 1; i++) { fx.mode = "serve"; fx.queue = (effectLog[i] || []).slice(); + emits.entries = []; injectGrants(perCell[i]); try { await cells[i](); @@ -377,6 +501,7 @@ export default class ReplRunner extends WorkerEntrypoint { fx.recorded = []; logs.entries = []; logs.dropped = 0; + emits.entries = []; injectGrants(current); let value; try { @@ -387,6 +512,9 @@ export default class ReplRunner extends WorkerEntrypoint { phase: "cell", error: describeError(error, undefined), logs: snapshotLogs(), + // A failing cell's emits still ship (they narrate the failure); + // the cell itself is never committed. + results: emits.entries.slice(), }; } @@ -394,7 +522,9 @@ export default class ReplRunner extends WorkerEntrypoint { ok: true, effects: fx.recorded, logs: snapshotLogs(), - results: [], + // Emitted entries first; an unclonable completion value's rendering + // appends after them (Jupyter display order). + results: emits.entries.slice(), }; try { // A capability handle is not a value — it structured-clones as an diff --git a/packages/computer/src/repl/session.ts b/packages/computer/src/repl/session.ts index ca86bebc..01ae944b 100644 --- a/packages/computer/src/repl/session.ts +++ b/packages/computer/src/repl/session.ts @@ -73,7 +73,7 @@ interface RunOutcome { error?: ReplExecutionError; effects?: ReplEffect[]; logs?: { entries: ReplLogEntry[]; dropped?: number }; - results?: Array<{ text: string }>; + results?: Array<{ text: string; value?: unknown }>; hasValue?: boolean; value?: unknown; } @@ -191,7 +191,7 @@ export class ReplSession { return { code, logs: outcome.logs ?? EMPTY_LOGS(), - results: [], + results: outcome.results ?? [], error: outcome.error ?? { name: "Error", message: "REPL evaluation failed." }, executionCount, }; diff --git a/packages/computer/src/repl/tool.test.ts b/packages/computer/src/repl/tool.test.ts new file mode 100644 index 00000000..fcca571e --- /dev/null +++ b/packages/computer/src/repl/tool.test.ts @@ -0,0 +1,169 @@ +// Unit tests for the framework-agnostic `js` tool definition: generated +// description, JSON Schema shape, session routing, input validation, and +// the shared text renderer. Real eval behavior is covered by the workerd +// integration suites; here the workspace is a recording fake. + +import { describe, expect, it } from "vitest"; + +import { capability, fetchCapability } from "./capability.js"; +import { createJsToolDefinition, type JsToolOptions, renderJsResultText } from "./tool.js"; +import type { ReplExecutionResult } from "./types.js"; + +interface ReplCall { + name: string; + options: Record; + code?: string; +} + +function fakeWorkspace(calls: ReplCall[]) { + return { + repl(name: string, options: Record) { + const call: ReplCall = { name, options }; + calls.push(call); + return { + eval(code: string): Promise { + call.code = code; + return Promise.resolve({ + code, + value: `ran in ${name}`, + logs: { entries: [] }, + results: [], + executionCount: 1, + }); + }, + }; + }, + }; +} + +const LOADER = {} as JsToolOptions["loader"]; + +function makeTool(overrides: Partial = {}) { + const calls: ReplCall[] = []; + const definition = createJsToolDefinition({ + workspace: fakeWorkspace(calls), + loader: LOADER, + ...overrides, + }); + return { definition, calls }; +} + +describe("createJsToolDefinition", () => { + it("describes persistence, built-ins, and each grant with its description", () => { + const { definition } = makeTool({ + capabilities: { + weather: capability({ get: () => 1 }, { description: "Weather lookups" }), + fetch: fetchCapability({ allow: ["api.example.com"] }), + bare: capability(() => 0), + }, + }); + expect(definition.name).toBe("js"); + expect(definition.description).toContain("persistent"); + expect(definition.description).toContain("emit(value)"); + expect(definition.description).toContain('session "main"'); + expect(definition.description).toContain("weather (Weather lookups)"); + expect(definition.description).toContain("fetch (HTTP fetch restricted to: api.example.com)"); + expect(definition.description).toContain("bare"); + expect(definition.description).toContain('help("name")'); + }); + + it("says so when nothing is granted", () => { + const { definition } = makeTool(); + expect(definition.description).toContain("No capabilities are granted"); + expect(definition.description).toContain("help()"); + }); + + it("names the configured default session in description and schema", () => { + const { definition } = makeTool({ + defaultSession: "scratch", + capabilities: { weather: capability({ get: () => 1 }) }, + }); + expect(definition.description).toContain('session "scratch"'); + expect(JSON.stringify(definition.inputSchema)).toContain('\\"scratch\\"'); + }); + + it("exposes a plain JSON Schema requiring only code", () => { + const { definition } = makeTool(); + expect(definition.inputSchema.type).toBe("object"); + expect(definition.inputSchema.required).toEqual(["code"]); + expect(definition.inputSchema.additionalProperties).toBe(false); + const code = definition.inputSchema.properties.code as { type: string }; + const sessionName = definition.inputSchema.properties.sessionName as { type: string }; + expect(code.type).toBe("string"); + expect(sessionName.type).toBe("string"); + }); + + it("routes to the default session and forwards grants and limits", async () => { + const weather = capability({ get: () => 1 }); + const { definition, calls } = makeTool({ + capabilities: { weather }, + timeoutMs: 5_000, + maxEffectBytes: 1_024, + }); + const result = await definition.run({ code: "1 + 1" }); + expect(result.value).toBe("ran in main"); + expect(calls).toHaveLength(1); + expect(calls[0].name).toBe("main"); + expect(calls[0].code).toBe("1 + 1"); + expect(calls[0].options).toEqual({ + loader: LOADER, + capabilities: { weather }, + timeoutMs: 5_000, + maxEffectBytes: 1_024, + }); + }); + + it("routes sessionName overrides and omits unset limits", async () => { + const { definition, calls } = makeTool(); + const result = await definition.run({ code: "2", sessionName: "notes" }); + expect(result.value).toBe("ran in notes"); + expect(calls[0].name).toBe("notes"); + expect(calls[0].options).toEqual({ loader: LOADER, capabilities: {} }); + }); + + it("rejects caller mistakes with TypeErrors", async () => { + const { definition } = makeTool(); + await expect(definition.run({ code: 1 as unknown as string })).rejects.toThrow(TypeError); + await expect(definition.run({ code: "1", sessionName: "" })).rejects.toThrow(TypeError); + expect(() => createJsToolDefinition({ + workspace: fakeWorkspace([]), + loader: LOADER, + defaultSession: "", + })).toThrow(TypeError); + }); +}); + +describe("renderJsResultText", () => { + const base = { code: "", logs: { entries: [] }, results: [], executionCount: 1 }; + + it("renders logs, emits, and value in execution order", () => { + const text = renderJsResultText({ + ...base, + value: { total: 3 }, + logs: { entries: [{ level: "warn" as const, text: "careful" }], dropped: 2 }, + results: [{ text: "emitted-one", value: 1 }], + }); + expect(text).toBe( + '[warn] careful\n(2 more log entries dropped)\nemitted-one\nvalue: {"total":3}', + ); + }); + + it("renders structured errors with kind and traceback", () => { + const text = renderJsResultText({ + ...base, + error: { + name: "StaleLeaseError", + message: "handle died", + kind: "stale-lease", + traceback: "at cell:1", + }, + }); + expect(text).toBe("StaleLeaseError [stale-lease]: handle died\nat cell:1"); + }); + + it("renders undefined and unstringifiable values honestly", () => { + expect(renderJsResultText({ ...base, value: undefined })).toBe("value: undefined"); + expect(renderJsResultText({ ...base, value: 10n })).toBe('value: "10n"'); + expect(renderJsResultText(base)).toBe("value: undefined"); + }); +}); diff --git a/packages/computer/src/repl/tool.ts b/packages/computer/src/repl/tool.ts new file mode 100644 index 00000000..4e087625 --- /dev/null +++ b/packages/computer/src/repl/tool.ts @@ -0,0 +1,196 @@ +// Framework-agnostic `js` tool definition. +// +// A model-facing tool is four things — name, description, input schema, +// execute — and every agent framework (AI SDK, pi, MCP servers, +// OpenAI/LangChain-style registries) wraps that same quartet, because the +// shape is pinned from below by the providers' function-calling APIs. This +// definition carries the quartet with zero framework dependencies: the +// schema is plain JSON Schema (every framework's lingua franca) and the +// output is the session's own `ReplExecutionResult`. Shims adapt it in a +// few lines — see src/tools/js.ts for the AI SDK one; a pi or MCP wrapper +// is the same handful of lines around `run()` and `renderJsResultText()`. + +import type { WorkspaceRuntimeLoader } from "../runtime/types.js"; +import type { ReplCapability } from "./capability.js"; +import type { ReplExecutionResult } from "./types.js"; + +export interface JsToolInput { + code: string; + sessionName?: string; +} + +/** The slice of a session the tool drives. `ReplSession` satisfies it. */ +export interface JsToolSessionLike { + eval(code: string): Promise; +} + +/** The slice of `Workspace` the tool needs. */ +export interface JsToolWorkspaceLike { + repl( + name: string, + options: { + loader: WorkspaceRuntimeLoader; + capabilities?: Record; + timeoutMs?: number; + maxEffectBytes?: number; + }, + ): JsToolSessionLike; +} + +export interface JsToolOptions { + workspace: JsToolWorkspaceLike; + loader: WorkspaceRuntimeLoader; + /** + * Capabilities granted to every session this tool touches, by global + * name. Grants are attach-time: each call re-attaches exactly this set, + * and the generated tool description lists it. + */ + capabilities?: Record; + /** Session used when the model omits `sessionName`. Default "main". */ + defaultSession?: string; + timeoutMs?: number; + maxEffectBytes?: number; +} + +export interface JsToolDefinition { + name: "js"; + /** Generated from the grant set — regenerate by recreating the tool. */ + description: string; + /** + * Plain JSON Schema for {@link JsToolInput}. Typed with concrete + * literals so it satisfies stricter framework types (e.g. the AI SDK's + * JSONSchema7) without this module depending on any of them. + */ + inputSchema: { + type: "object"; + properties: { + code: { type: "string"; description: string }; + sessionName: { type: "string"; description: string }; + }; + required: ["code"]; + additionalProperties: false; + }; + /** + * Evaluate one cell. Structured failures (including the cell's own + * errors) return as `result.error` — this only throws for caller + * mistakes like a non-string `code`. `signal` is accepted for shim + * uniformity; a running cell is bounded by its timeout rather than + * cancelled mid-flight, so aborting affects only the caller's await. + */ + run(input: JsToolInput, options?: { signal?: AbortSignal }): Promise; +} + +export function createJsToolDefinition(options: JsToolOptions): JsToolDefinition { + const defaultSession = options.defaultSession ?? "main"; + if (typeof defaultSession !== "string" || defaultSession === "") { + throw new TypeError("createJsToolDefinition: defaultSession must be a non-empty string."); + } + const capabilities = options.capabilities ?? {}; + + return { + name: "js", + description: jsToolDescription(defaultSession, capabilities), + inputSchema: { + type: "object", + properties: { + code: { + type: "string", + description: + "JavaScript source for one cell. Top-level await works; the last expression (or an explicit return) is the cell's value.", + }, + sessionName: { + type: "string", + description: `Named session to evaluate in. Omit for "${defaultSession}". Each session is its own persistent state.`, + }, + }, + required: ["code"], + additionalProperties: false, + }, + async run(input) { + if (typeof input?.code !== "string") { + throw new TypeError("js tool: `code` must be a string of JavaScript source."); + } + const sessionName = input.sessionName ?? defaultSession; + if (typeof sessionName !== "string" || sessionName === "") { + throw new TypeError("js tool: `sessionName` must be a non-empty string."); + } + const session = options.workspace.repl(sessionName, { + loader: options.loader, + capabilities, + ...(options.timeoutMs !== undefined ? { timeoutMs: options.timeoutMs } : {}), + ...(options.maxEffectBytes !== undefined + ? { maxEffectBytes: options.maxEffectBytes } + : {}), + }); + return session.eval(input.code); + }, + }; +} + +function jsToolDescription( + defaultSession: string, + capabilities: Record, +): string { + const intro = + "Run JavaScript in a persistent session. Variables, functions, and classes " + + "survive across calls — including restarts — so build state up instead of " + + "re-sending it. Top-level `await` works; the value of the last expression " + + "(or an explicit `return`) comes back with any console logs. Use " + + "`emit(value)` to publish extra structured results."; + const names = Object.keys(capabilities); + if (names.length === 0) { + return ( + `${intro} No capabilities are granted — no network or filesystem; ` + + "pure JavaScript with durable state. `help()` in a cell lists the built-ins." + ); + } + const grants = names + .map((name) => { + const description = capabilities[name].meta.description; + return description === undefined ? name : `${name} (${description})`; + }) + .join(", "); + return ( + `${intro} Capabilities in session "${defaultSession}": ${grants}. ` + + 'Call `help("name")` in a cell for full docs on any of them.' + ); +} + +/** + * Render an eval result as compact text, in execution-narrative order: + * logs, emitted results, then the completion value or error. One shared + * renderer so every framework shim shows the model the same thing. + */ +export function renderJsResultText(result: ReplExecutionResult): string { + const lines: string[] = []; + for (const entry of result.logs.entries) { + lines.push(`[${entry.level}] ${entry.text}`); + } + if (result.logs.dropped !== undefined && result.logs.dropped > 0) { + lines.push(`(${result.logs.dropped} more log entries dropped)`); + } + for (const entry of result.results) { + lines.push(entry.text); + } + if (result.error) { + const kind = result.error.kind === undefined ? "" : ` [${result.error.kind}]`; + lines.push(`${result.error.name}${kind}: ${result.error.message}`); + if (result.error.traceback !== undefined) lines.push(result.error.traceback); + } else if ("value" in result) { + lines.push(`value: ${stringifyValue(result.value)}`); + } + return lines.length === 0 ? "value: undefined" : lines.join("\n"); +} + +function stringifyValue(value: unknown): string { + if (value === undefined) return "undefined"; + try { + const json = JSON.stringify(value, (_key, v: unknown) => + typeof v === "bigint" ? `${v}n` : v, + ); + if (json !== undefined) return json; + } catch { + // Cycles and other non-JSON values fall through to String(). + } + return String(value); +} diff --git a/packages/computer/src/repl/types.ts b/packages/computer/src/repl/types.ts index 11c36215..be7e0983 100644 --- a/packages/computer/src/repl/types.ts +++ b/packages/computer/src/repl/types.ts @@ -11,6 +11,13 @@ export interface ReplResultData { /** Inspect-style text rendering of an output. */ text: string; + /** + * The output's structured value (an `emit(value)` argument, cloned at + * emit time). Present when the value can cross the boundary and fits + * the per-entry ceiling; absent for pure renderings — unclonable + * values, capability handles, and oversized emits. + */ + value?: unknown; } export type ReplLogLevel = "log" | "info" | "debug" | "warn" | "error"; @@ -82,7 +89,13 @@ export interface ReplExecutionResult { * preserved. `dropped` counts entries discarded past the cap. */ logs: { entries: ReplLogEntry[]; dropped?: number }; - /** Display renderings (e.g. of an unclonable value). */ + /** + * Display outputs, in order: the new cell's `emit(value)` entries, + * then a rendering of the completion value when it could not ship on + * `value`. Replayed cells' emits never reappear here — like console + * output, they belong to the eval that ran them. A failing cell's + * emits still arrive (they narrate the failure). + */ results: ReplResultData[]; error?: ReplExecutionError; /** 1-based cell sequence (the would-be sequence for failed cells). */ diff --git a/packages/computer/src/tools/index.ts b/packages/computer/src/tools/index.ts index 9bad4745..0ab4b9b6 100644 --- a/packages/computer/src/tools/index.ts +++ b/packages/computer/src/tools/index.ts @@ -10,6 +10,7 @@ export { export { createDeleteTool, type DeleteToolOptions } from "./fs/delete.js"; export { createEditTool, type EditToolOptions } from "./fs/edit.js"; export { createFindTool, type FindToolOptions } from "./fs/find.js"; +export { createJsTool, type JsToolInput, type JsToolOptions } from "./js.js"; export { createGrepTool, type GrepToolOptions } from "./fs/grep.js"; export { createListTool, type ListToolOptions } from "./fs/list.js"; export { createReadTool, type LineTruncation, type ReadToolOptions } from "./fs/read.js"; diff --git a/packages/computer/src/tools/js.test.ts b/packages/computer/src/tools/js.test.ts new file mode 100644 index 00000000..f04c01e7 --- /dev/null +++ b/packages/computer/src/tools/js.test.ts @@ -0,0 +1,54 @@ +// The AI SDK shim is a thin dress over createJsToolDefinition — these +// tests pin the seams: description and schema pass through, and execute +// routes to the definition's run(). + +import { describe, expect, it } from "vitest"; + +import type { JsToolOptions } from "../repl/tool.js"; +import type { ReplExecutionResult } from "../repl/types.js"; +import { createJsTool } from "./js.js"; + +function fakeWorkspace(calls: Array<{ name: string; code: string }>) { + return { + repl(name: string) { + return { + eval(code: string): Promise { + calls.push({ name, code }); + return Promise.resolve({ + code, + value: 42, + logs: { entries: [] }, + results: [], + executionCount: 1, + }); + }, + }; + }, + }; +} + +describe("createJsTool (AI SDK shim)", () => { + it("passes the generated description and JSON Schema through", () => { + const jsTool = createJsTool({ + workspace: fakeWorkspace([]), + loader: {} as JsToolOptions["loader"], + }); + expect(jsTool.description).toContain("persistent"); + const schema = jsTool.inputSchema as { jsonSchema?: { required?: string[] } }; + expect(schema.jsonSchema?.required).toEqual(["code"]); + }); + + it("executes through the definition", async () => { + const calls: Array<{ name: string; code: string }> = []; + const jsTool = createJsTool({ + workspace: fakeWorkspace(calls), + loader: {} as JsToolOptions["loader"], + }); + const result = (await jsTool.execute?.( + { code: "6 * 7", sessionName: "notes" }, + { toolCallId: "t1", messages: [], abortSignal: new AbortController().signal }, + )) as ReplExecutionResult; + expect(result.value).toBe(42); + expect(calls).toEqual([{ name: "notes", code: "6 * 7" }]); + }); +}); diff --git a/packages/computer/src/tools/js.ts b/packages/computer/src/tools/js.ts new file mode 100644 index 00000000..7697282e --- /dev/null +++ b/packages/computer/src/tools/js.ts @@ -0,0 +1,26 @@ +// AI SDK shim for the `js` tool. The tool itself is framework-agnostic — +// `createJsToolDefinition` in src/repl/tool.ts carries the description, +// JSON Schema, and executor with zero framework dependencies. This file is +// the few lines that dress it as an AI SDK tool; a pi or MCP wrapper is +// the same pattern around `run()` (see renderJsResultText for text-only +// frameworks). + +import { jsonSchema, type Tool, tool } from "ai"; + +import { + createJsToolDefinition, + type JsToolInput, + type JsToolOptions, +} from "../repl/tool.js"; +import type { ReplExecutionResult } from "../repl/types.js"; + +export type { JsToolInput, JsToolOptions }; + +export function createJsTool(options: JsToolOptions): Tool { + const definition = createJsToolDefinition(options); + return tool({ + description: definition.description, + inputSchema: jsonSchema(definition.inputSchema), + execute: (input, { abortSignal }) => definition.run(input, { signal: abortSignal }), + }); +} diff --git a/packages/computer/tests/repl-builtins.test.ts b/packages/computer/tests/repl-builtins.test.ts new file mode 100644 index 00000000..befd2eff --- /dev/null +++ b/packages/computer/tests/repl-builtins.test.ts @@ -0,0 +1,231 @@ +// Durable REPL sessions — in-session built-in tests: help() and emit(). +// +// Runs against real workerd. The discipline under test: built-ins are not +// capabilities and are never recorded. help() is a pure function of the +// cell's injected grant shapes (replayed cells see their recorded +// snapshot, so committed help() calls survive revocation and restart), +// and emit() fills a per-cell display buffer — replayed cells' emits are +// discarded exactly like their console output. + +import { env } from "cloudflare:test"; +import { describe, expect, it } from "vitest"; + +import type { ReplHostDO } from "./repl-worker.js"; + +let sessionCounter = 0; + +function uniqueSession(): string { + sessionCounter += 1; + return `builtins-session-${sessionCounter}`; +} + +function host() { + const id = env.HOST.newUniqueId(); + return env.HOST.get(id); +} + +describe("REPL help()", () => { + it("overview lists built-ins and states when nothing is granted", async () => { + const stub = host(); + const result = await stub.replEval(uniqueSession(), "help()"); + expect(result.error).toBeUndefined(); + const text = result.value as string; + expect(text).toContain("help("); + expect(text).toContain("emit("); + expect(text).toContain("No capabilities are granted"); + }); + + it("overview lists each grant with its description", async () => { + const stub = host(); + const result = await stub.replEvalWith(["weather", "configV1"], uniqueSession(), "help()"); + expect(result.error).toBeUndefined(); + const text = result.value as string; + expect(text).toContain("weather — Weather lookups"); + expect(text).toContain("config"); + expect(text).not.toContain("No capabilities are granted"); + }); + + it("help(name) shows methods with grantor docs", async () => { + const stub = host(); + const result = await stub.replEvalWith(["weather"], uniqueSession(), 'help("weather")'); + expect(result.error).toBeUndefined(); + const text = result.value as string; + expect(text).toContain("Weather lookups"); + expect(text).toContain("get(city) → { city, temp, asOf }"); + expect(text).toContain("weather.flaky("); + }); + + it("help(name) walks nested children and data snapshots", async () => { + const stub = host(); + const result = await stub.replEvalWith(["api"], uniqueSession(), 'help("api")'); + expect(result.error).toBeUndefined(); + const text = result.value as string; + expect(text).toContain("api.users.list("); + expect(text).toContain('api.version = "2.0"'); + }); + + it("help(name) says so for opaque surfaces", async () => { + const stub = host(); + const result = await stub.replEvalWith(["stub"], uniqueSession(), 'help("stub")'); + expect(result.error).toBeUndefined(); + expect(result.value as string).toContain("opaque"); + }); + + it("help(unknown) names the grants that do exist", async () => { + const stub = host(); + const result = await stub.replEvalWith(["weather"], uniqueSession(), 'help("nope")'); + expect(result.error).toBeUndefined(); + const text = result.value as string; + expect(text).toContain('No capability named "nope"'); + expect(text).toContain("weather"); + }); + + it("committed help() replays from its snapshot after revocation and restart", async () => { + const stub = host(); + const session = uniqueSession(); + + const first = await stub.replEvalWith( + ["weather"], + session, + 'const w = help("weather"); w.includes("get(city)")', + ); + expect(first.error).toBeUndefined(); + expect(first.value).toBe(true); + + // Revoke everything (attach with no grants) across a restart. Cell 1 + // must replay against its recorded shape snapshot — no divergence — + // while the new cell's help() reflects the now-empty attachment. + await stub.restart(); + const second = await stub.replEval(session, 'w.length > 0 && !help().includes("weather")'); + expect(second.error).toBeUndefined(); + expect(second.value).toBe(true); + }); +}); + +describe("js tool definition (end to end)", () => { + it("describes its grants and evaluates in real durable sessions", async () => { + const stub = host(); + + const description = await stub.jsToolDescription(["weather"]); + expect(description).toContain('session "main"'); + expect(description).toContain("weather (Weather lookups)"); + + // Default session: state persists across calls and grants work. + const first = await stub.jsToolRun( + ["weather"], + 'const t = (await weather.get("oslo")).temp; t', + ); + expect(first.error).toBeUndefined(); + expect(first.value).toBe(21.5); + const second = await stub.jsToolRun(["weather"], "t * 2"); + expect(second.error).toBeUndefined(); + expect(second.value).toBe(43); + expect(await stub.counts()).toEqual({ "weather.get": 1 }); + + // A named session is its own state. + const other = await stub.jsToolRun(["weather"], "typeof t", "scratch"); + expect(other.value).toBe("undefined"); + expect(other.executionCount).toBe(1); + }); +}); + +describe("REPL emit()", () => { + it("delivers emitted values in order, structured plus rendered", async () => { + const stub = host(); + const result = await stub.replEval( + uniqueSession(), + 'emit(1); emit("two"); console.log("between"); emit({ d: new Date(5) }); "done"', + ); + expect(result.error).toBeUndefined(); + expect(result.value).toBe("done"); + expect(result.results).toHaveLength(3); + expect(result.results[0].value).toBe(1); + expect(result.results[1].value).toBe("two"); + const emitted = result.results[2].value as { d: Date }; + expect(emitted.d).toBeInstanceOf(Date); + expect(emitted.d.getTime()).toBe(5); + for (const entry of result.results) expect(entry.text.length).toBeGreaterThan(0); + expect(result.logs.entries).toEqual([{ level: "log", text: "between" }]); + }); + + it("returns the emitted value for inline use", async () => { + const stub = host(); + const result = await stub.replEval(uniqueSession(), "const r = emit(7); r + 1"); + expect(result.error).toBeUndefined(); + expect(result.value).toBe(8); + }); + + it("ships unclonable emits as rendering only", async () => { + const stub = host(); + const result = await stub.replEval(uniqueSession(), "emit(() => 1); 0"); + expect(result.error).toBeUndefined(); + expect(result.results).toHaveLength(1); + expect("value" in result.results[0]).toBe(false); + expect(result.results[0].text).toContain("Function"); + }); + + it("ships capability handles as rendering only", async () => { + const stub = host(); + const result = await stub.replEvalWith(["weather"], uniqueSession(), 'emit(weather); "ok"'); + expect(result.error).toBeUndefined(); + expect(result.results).toHaveLength(1); + expect("value" in result.results[0]).toBe(false); + expect(result.results[0].text).toContain("capability handle"); + }); + + it("omits oversized structured values and says where to put them", async () => { + const stub = host(); + const result = await stub.replEval(uniqueSession(), 'emit("x".repeat(300000)); "ok"'); + expect(result.error).toBeUndefined(); + expect(result.results).toHaveLength(1); + const entry = result.results[0]; + expect("value" in entry).toBe(false); + expect(entry.text).toContain("omitted"); + expect(entry.text).toContain("workspace file"); + expect(entry.text.length).toBeLessThan(10_000); + }); + + it("suppresses replayed cells' emits, live and across restart", async () => { + const stub = host(); + const session = uniqueSession(); + + const first = await stub.replEval(session, 'emit("a"); 1'); + expect(first.results).toHaveLength(1); + + const second = await stub.replEval(session, "2"); + expect(second.error).toBeUndefined(); + expect(second.results).toHaveLength(0); + + await stub.restart(); + const third = await stub.replEval(session, "3"); + expect(third.error).toBeUndefined(); + expect(third.results).toHaveLength(0); + expect(third.executionCount).toBe(3); + }); + + it("delivers emits from a failing cell without committing it", async () => { + const stub = host(); + const session = uniqueSession(); + + const failed = await stub.replEval(session, 'emit("before"); throw new Error("boom")'); + expect(failed.error?.message).toBe("boom"); + expect(failed.results).toHaveLength(1); + expect(failed.results[0].value).toBe("before"); + + const after = await stub.replEval(session, '"after"'); + expect(after.executionCount).toBe(failed.executionCount); + }); + + it("throws past the per-cell emit cap instead of dropping silently", async () => { + const stub = host(); + const session = uniqueSession(); + + const failed = await stub.replEval(session, "for (let i = 0; i < 1001; i++) emit(i);"); + expect(failed.error?.message).toContain("1000"); + expect(failed.error?.message).toContain("emit"); + + const after = await stub.replEval(session, '"after"'); + expect(after.error).toBeUndefined(); + expect(after.executionCount).toBe(failed.executionCount); + }); +}); diff --git a/packages/computer/tests/repl-worker.ts b/packages/computer/tests/repl-worker.ts index 99448a50..c3b1d396 100644 --- a/packages/computer/tests/repl-worker.ts +++ b/packages/computer/tests/repl-worker.ts @@ -12,7 +12,13 @@ import type { ReplEvalOptions, ReplExecutionResult, } from "../src/index.js"; -import { capability, fetchCapability, Workspace, workspaceFs } from "../src/index.js"; +import { + capability, + createJsToolDefinition, + fetchCapability, + Workspace, + workspaceFs, +} from "../src/index.js"; export interface Env { HOST: DurableObjectNamespace; @@ -237,6 +243,29 @@ export class ReplHostDO extends DurableObject { return this.#ws().repl(session, { loader: this.env.LOADER }).eval(code, options); } + // End-to-end path for the framework-agnostic js tool: the real + // Workspace must satisfy JsToolWorkspaceLike, and run() must route to + // real durable sessions. + async jsToolRun( + fixtures: string[], + code: string, + sessionName?: string, + ): Promise { + return this.#jsTool(fixtures).run({ code, ...(sessionName === undefined ? {} : { sessionName }) }); + } + + jsToolDescription(fixtures: string[]): string { + return this.#jsTool(fixtures).description; + } + + #jsTool(fixtures: string[]) { + return createJsToolDefinition({ + workspace: this.#ws(), + loader: this.env.LOADER, + capabilities: this.#fixtures(fixtures), + }); + } + // Simulate DO eviction: drop every in-memory object. Storage survives. restart(): void { this.#workspace = undefined; diff --git a/packages/computer/vitest.config.repl.ts b/packages/computer/vitest.config.repl.ts index cb03d631..99235723 100644 --- a/packages/computer/vitest.config.repl.ts +++ b/packages/computer/vitest.config.repl.ts @@ -10,7 +10,7 @@ export default defineConfig({ ], test: { globals: true, - include: ["tests/repl.test.ts", "tests/repl-capabilities.test.ts"], + include: ["tests/repl.test.ts", "tests/repl-capabilities.test.ts", "tests/repl-builtins.test.ts"], testTimeout: 60_000, }, }); From 514c0da3bf50697d65038707e9c99d0c90822361 Mon Sep 17 00:00:00 2001 From: Ben Reitz Date: Fri, 11 Sep 2026 14:50:03 +0100 Subject: [PATCH 2/3] fix: allow workspace runtime writeFile to root-level paths MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit dofs deliberately EEXISTs mkdir("/") even with recursive, but writeFile unconditionally mkdir'd the parent — so writing any file directly under the workspace root failed. Pre-existing on main since f1cf132c; caught while exercising workspaceFs through the durable REPL playground. Regression covered in the fs capability replay test. --- packages/computer/src/runtime/capability.ts | 4 +++- packages/computer/tests/repl-capabilities.test.ts | 9 ++++++--- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/packages/computer/src/runtime/capability.ts b/packages/computer/src/runtime/capability.ts index 96e8a500..b8833ca6 100644 --- a/packages/computer/src/runtime/capability.ts +++ b/packages/computer/src/runtime/capability.ts @@ -135,7 +135,9 @@ export class WorkspaceRuntimeCapability { this.#requireWrite(); const resolved = await this.#resolveSafe(path, true); const parent = resolved.slice(0, resolved.lastIndexOf("/")) || this.#root; - await this.#fs.mkdir(parent, { recursive: true }); + // dofs treats mkdir("/") as EEXIST even with recursive — root always + // exists — so writes to root-level files must not try to create it. + if (parent !== "/") await this.#fs.mkdir(parent, { recursive: true }); await this.#assertSafeComponents(parent, false); await this.#fs.writeFile(resolved, content); } diff --git a/packages/computer/tests/repl-capabilities.test.ts b/packages/computer/tests/repl-capabilities.test.ts index fc23293f..f818f5b1 100644 --- a/packages/computer/tests/repl-capabilities.test.ts +++ b/packages/computer/tests/repl-capabilities.test.ts @@ -340,11 +340,14 @@ describe("REPL capabilities", () => { session, `await fs.mkdir("/notes"); await fs.writeFile("/notes/log.txt", "alpha"); + // Regression: root-level writes must not mkdir("/") (dofs EEXISTs it). + await fs.writeFile("/top.txt", "root-level"); + await fs.readFile("/top.txt"); await fs.readFile("/notes/log.txt")`, ); expect(first.error).toBeUndefined(); expect(first.value).toBe("alpha"); - expect(await stub.counts()).toMatchObject({ "fs.writeFile": 1, "fs.readFile": 1 }); + expect(await stub.counts()).toMatchObject({ "fs.writeFile": 2, "fs.readFile": 2 }); const second = await stub.replEvalWith( ["fs"], @@ -354,14 +357,14 @@ describe("REPL capabilities", () => { await fs.readFile("/notes/log.txt")`, ); expect(second.value).toBe("alpha+beta"); - expect(await stub.counts()).toMatchObject({ "fs.writeFile": 2, "fs.readFile": 3 }); + expect(await stub.counts()).toMatchObject({ "fs.writeFile": 3, "fs.readFile": 4 }); // Replay after eviction re-fires nothing: same content, same counters // (+1 read for the new cell's own live read). await stub.restart(); const third = await stub.replEvalWith(["fs"], session, `await fs.readFile("/notes/log.txt")`); expect(third.value).toBe("alpha+beta"); - expect(await stub.counts()).toMatchObject({ "fs.writeFile": 2, "fs.readFile": 4 }); + expect(await stub.counts()).toMatchObject({ "fs.writeFile": 3, "fs.readFile": 5 }); }); it("rejects oversized capability results without committing or truncating", async () => { From 5784d23210f10b653721461f6ad7b3c9848d6fa5 Mon Sep 17 00:00:00 2001 From: Ben Reitz Date: Fri, 11 Sep 2026 14:50:04 +0100 Subject: [PATCH 3/3] fix: say capability calls are async in help() and the js tool description MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit First minutes of real playground use produced exactly the predicted model mistake: calling a capability without await and failing on the returned Promise. help() now renders every method as `await cap.m(…)`, the overview states calls are asynchronous, and the generated tool description says so too. --- packages/computer/src/repl/runner.ts | 7 ++++--- packages/computer/src/repl/tool.test.ts | 1 + packages/computer/src/repl/tool.ts | 1 + packages/computer/tests/repl-builtins.test.ts | 2 ++ 4 files changed, 8 insertions(+), 3 deletions(-) diff --git a/packages/computer/src/repl/runner.ts b/packages/computer/src/repl/runner.ts index c3a2bc7e..04fe1360 100644 --- a/packages/computer/src/repl/runner.ts +++ b/packages/computer/src/repl/runner.ts @@ -409,6 +409,7 @@ function helpText(name) { const d = shapes[n] && shapes[n].description; lines.push(" " + n + (d ? " — " + d : "")); } + lines.push("Capability calls are asynchronous — always await them."); lines.push("help(\\"name\\") shows a capability's methods and docs."); } return lines.join("\\n"); @@ -429,15 +430,15 @@ function helpText(name) { // surfaces), data snapshots with their current values, and children. function renderShapeHelp(lines, path, docPath, shape, docs) { if (shape.opaque) { - lines.push(" " + path + ".(…) — surface unknown (opaque remote stub): call any method it supports"); + lines.push(" await " + path + ".(…) — surface unknown (opaque remote stub): call any method it supports"); return; } if (shape.callable) { - lines.push(" " + path + "(…) — callable directly" + (docs[docPath] ? ": " + docs[docPath] : "")); + lines.push(" await " + path + "(…) — callable directly" + (docs[docPath] ? ": " + docs[docPath] : "")); } for (const m of shape.methods || []) { const dk = docPath === "" ? m : docPath + "." + m; - lines.push(" " + path + "." + m + "(…)" + (docs[dk] ? " — " + docs[dk] : "")); + lines.push(" await " + path + "." + m + "(…)" + (docs[dk] ? " — " + docs[dk] : "")); } for (const k of Object.keys(shape.data || {})) { let rendered; diff --git a/packages/computer/src/repl/tool.test.ts b/packages/computer/src/repl/tool.test.ts index fcca571e..d28d09c8 100644 --- a/packages/computer/src/repl/tool.test.ts +++ b/packages/computer/src/repl/tool.test.ts @@ -64,6 +64,7 @@ describe("createJsToolDefinition", () => { expect(definition.description).toContain("weather (Weather lookups)"); expect(definition.description).toContain("fetch (HTTP fetch restricted to: api.example.com)"); expect(definition.description).toContain("bare"); + expect(definition.description).toContain("async — always `await`"); expect(definition.description).toContain('help("name")'); }); diff --git a/packages/computer/src/repl/tool.ts b/packages/computer/src/repl/tool.ts index 4e087625..42350918 100644 --- a/packages/computer/src/repl/tool.ts +++ b/packages/computer/src/repl/tool.ts @@ -152,6 +152,7 @@ function jsToolDescription( .join(", "); return ( `${intro} Capabilities in session "${defaultSession}": ${grants}. ` + + "Capability calls are async — always `await` them. " + 'Call `help("name")` in a cell for full docs on any of them.' ); } diff --git a/packages/computer/tests/repl-builtins.test.ts b/packages/computer/tests/repl-builtins.test.ts index befd2eff..33a3f884 100644 --- a/packages/computer/tests/repl-builtins.test.ts +++ b/packages/computer/tests/repl-builtins.test.ts @@ -42,6 +42,7 @@ describe("REPL help()", () => { const text = result.value as string; expect(text).toContain("weather — Weather lookups"); expect(text).toContain("config"); + expect(text).toContain("asynchronous — always await"); expect(text).not.toContain("No capabilities are granted"); }); @@ -52,6 +53,7 @@ describe("REPL help()", () => { const text = result.value as string; expect(text).toContain("Weather lookups"); expect(text).toContain("get(city) → { city, temp, asOf }"); + expect(text).toContain("await weather.get("); expect(text).toContain("weather.flaky("); });