diff --git a/.changeset/exec-tool-single-backend.md b/.changeset/exec-tool-single-backend.md new file mode 100644 index 00000000..701465bc --- /dev/null +++ b/.changeset/exec-tool-single-backend.md @@ -0,0 +1,7 @@ +--- +"@cloudflare/computer": minor +--- + +The `exec` tool offers only the arguments that can work. With one backend there is no `backend` argument, the tool always runs there, `defaultBackend` becomes optional, and the description talks about what that backend does rather than how to choose one. `input` appears only when a configured backend accepts it. + +Each backend's entry now adds what the backend says about itself, read through `workspace.runtime.describe(id)`. For `WorkerJavaScriptBackend` that is its source language and every module code can import, so `shell: { backends: { "worker-javascript": {} } }` is enough and the module list the model reads cannot drift from `modules`. A backend `description` is required only for a backend that does not describe itself. diff --git a/docs/09_tool_interface.md b/docs/09_tool_interface.md index a8629a1c..181a2ba7 100644 --- a/docs/09_tool_interface.md +++ b/docs/09_tool_interface.md @@ -55,7 +55,16 @@ export class Agent { Pass the returned AI SDK `ToolSet` to `generateText`, `streamText`, or an agent framework hook such as `getTools()`. -Pass `shell` only when the Workspace has matching backend ids: +Pass `shell` only when the Workspace has matching backend ids. With one backend, `exec` has no `backend` argument and always runs there: + +```ts +const tools = createAITools({ + workspace, + shell: { backends: { "worker-javascript": {} } }, +}); +``` + +With more than one, pass `defaultBackend` and the model picks a backend per call: ```ts const tools = createAITools({ @@ -244,7 +253,19 @@ The tool uses forced removal, so deleting a missing path succeeds. Set `recursiv ## `exec` -`exec` is opt-in. It calls `workspace.runtime.exec` with the configured backend and streams bounded output. Backend descriptions are included in the model-facing tool description, so describe capabilities and startup cost in plain language. +`exec` is opt-in. It calls `workspace.runtime.exec` with the configured backend and streams bounded output. + +Each backend's entry in the tool description joins two parts: the `description` you pass, and what the backend says about itself (`backend.description`, read through `workspace.runtime.describe(id)`). `WorkerJavaScriptBackend` describes its source language and every module code can import, so `{ "worker-javascript": {} }` is enough and the list stays in step with `modules`. A backend that does not describe itself needs a `description`. Describe capabilities and startup cost in plain language. + +The tool offers only the arguments that can work: + +| Backends | Arguments | +| --- | --- | +| One shell backend | `command`, `cwd`, `env` | +| One callable backend | `command`, `cwd`, `env`, `input` | +| More than one | `command`, `cwd`, `backend`, `env`, plus `input` when any is callable. `defaultBackend` is required. | + +A `backend` value the model sends anyway is dropped when only one backend is configured. The output still names the backend that ran. Wire this tool carefully: it executes arbitrary shell commands inside the configured backend. Treat its output as untrusted text when including it in later model input. Omit `shell` or use `readonly: true` when command execution is not part of the agent's job. diff --git a/docs/17_isolate_javascript.md b/docs/17_isolate_javascript.md index 7db07cf6..2fe49372 100644 --- a/docs/17_isolate_javascript.md +++ b/docs/17_isolate_javascript.md @@ -148,7 +148,7 @@ new WorkerJavaScriptBackend({ An import that is not built in, configured, or a relative Workspace path fails before the Worker is created. Caller source and durable files cannot shadow a configured or built-in module. -The backend describes its modules for a model in `backend.description`, which `workspace.runtime.describe(id)` returns. It is built from the same `modules` option the backend runs with, so it always matches what is installed: +The backend describes its modules for a model in `backend.description`, which `workspace.runtime.describe(id)` returns and the `exec` tool shows. It is built from the same `modules` option the backend runs with, so it always matches what is installed: ```text `command` is ECMAScript module source, run in an isolated JavaScript runtime. Relative imports resolve from `cwd` in the workspace. @@ -161,7 +161,7 @@ Modules code can import: - `ws:weather`: exports `forecast`. ``` -A factory adds its own text through a `description` property, as the prebuilt modules do. An object of functions is listed by its export names. +A factory adds its own text through a `description` property, as the prebuilt modules do. An object of functions is listed by its export names; say more about it in the `exec` tool's backend description if the model needs it. ### Built-in filesystem diff --git a/packages/computer/src/text-truncation.ts b/packages/computer/src/text-truncation.ts new file mode 100644 index 00000000..8100d812 --- /dev/null +++ b/packages/computer/src/text-truncation.ts @@ -0,0 +1,37 @@ +const encoder = new TextEncoder(); + +/** + * Cut text to at most `maxBytes` UTF-8 bytes without splitting a + * character, and say how much was left out. + * + * @param value - The text to cut. + * @param maxBytes - The largest number of UTF-8 bytes to keep. + * @returns The text unchanged when it fits, or its longest whole-character + * prefix followed by a `[truncated, N more bytes]` marker. + */ +export function truncateText(value: string, maxBytes: number): string { + const totalBytes = encoder.encode(value).byteLength; + if (totalBytes <= maxBytes) return value; + const { text, bytes } = utf8Prefix(value, maxBytes); + return `${text}\n\n[truncated, ${totalBytes - bytes} more bytes]`; +} + +/** + * The longest whole-character prefix of `value` that fits in `maxBytes` + * UTF-8 bytes. + * + * @param value - The text to cut. + * @param maxBytes - The largest number of UTF-8 bytes to keep. + * @returns The prefix and its size in bytes. + */ +export function utf8Prefix(value: string, maxBytes: number): { text: string; bytes: number } { + let bytes = 0; + let end = 0; + for (const char of value) { + const charBytes = encoder.encode(char).byteLength; + if (bytes + charBytes > maxBytes) break; + bytes += charBytes; + end += char.length; + } + return { text: value.slice(0, end), bytes }; +} diff --git a/packages/computer/src/tools/ai.test.ts b/packages/computer/src/tools/ai.test.ts index 5d283781..146c7c83 100644 --- a/packages/computer/src/tools/ai.test.ts +++ b/packages/computer/src/tools/ai.test.ts @@ -1,5 +1,8 @@ import { SQLiteTestStorage } from "@cloudflare/dofs/testing"; import { describe, expect, it } from "vitest"; +import { z } from "zod"; +import { WorkerJavaScriptBackend } from "../backends/worker-javascript/worker-javascript.js"; +import { createGitModule } from "../modules/git.js"; import type { WorkspaceRuntimeExecHandle, WorkspaceRuntimeResult } from "../runtime/types.js"; import { Workspace } from "../workspace.js"; import { @@ -85,6 +88,17 @@ function toolDescription(tool: unknown): string { return description; } +function inputSchema(tool: unknown): z.ZodType { + const schema = (tool as { inputSchema?: unknown }).inputSchema; + if (!(schema instanceof z.ZodType)) throw new Error("tool has no zod input schema"); + return schema; +} + +function inputProperties(tool: unknown): string[] { + const json = z.toJSONSchema(inputSchema(tool)) as { properties?: Record }; + return Object.keys(json.properties ?? {}).sort(); +} + function makeWorkspace(): Workspace { return new Workspace({ storage: new SQLiteTestStorage(), now: () => 1_700_000_000_000 }); } @@ -1697,7 +1711,165 @@ describe("createAITools callable exec", () => { }, }); - expect(toolDescription(tools.exec)).toContain("callable"); + expect(toolDescription(tools.exec)).toContain("Run code in the workspace"); + expect(toolDescription(tools.exec)).toContain("`result` field"); + expect(toolDescription(tools.exec)).toContain("JavaScript module runtime"); + }); + + it("adds what the backend says about itself after the caller's description", () => { + const workspace = { + runtime: { + async exec() { + throw new Error("not used"); + }, + isCallable: () => true, + describe: (id: string) => + id === "js" ? "Modules: `ws:weather` exports `forecast`." : undefined, + }, + }; + const withBoth = createAITools({ + workspace, + shell: { backends: { js: { description: "Use for data work." } } }, + }); + const withBackendOnly = createAITools({ workspace, shell: { backends: { js: {} } } }); + + expect(toolDescription(withBoth.exec)).toContain( + "Use for data work.\n\nModules: `ws:weather` exports `forecast`.", + ); + expect(toolDescription(withBackendOnly.exec)).toContain("`ws:weather` exports `forecast`"); + }); + + it("requires a description for a backend that does not describe itself", () => { + const workspace = { + runtime: { + async exec() { + throw new Error("not used"); + }, + }, + }; + + expect(() => createAITools({ workspace, shell: { backends: { shell: {} } } })).toThrow( + /does not describe itself/, + ); + }); +}); + +describe("createAITools exec against a real JavaScript backend", () => { + it("lists the backend's modules in the tool description", () => { + const workspace = new Workspace({ + storage: new SQLiteTestStorage(), + backends: [ + new WorkerJavaScriptBackend({ + loader: { load: () => ({ getEntrypoint: () => ({}) }) }, + modules: { + "tar-stream": "export default {};", + "ws:weather": { forecast: ([city]) => ({ city, sky: "clear" }) }, + "ws:git": createGitModule(), + }, + }), + ], + }); + const tools = createAITools({ + workspace, + shell: { backends: { "worker-javascript": {} } }, + }); + const description = toolDescription(tools.exec); + + expect(description).toContain("`node:fs/promises`"); + expect(description).toContain("- `tar-stream`: a bundled library."); + expect(description).toContain("- `ws:weather`: exports `forecast`."); + expect(description).toContain("- `ws:git`: The workspace's Git repository tools"); + expect(description).toContain("no direct network access"); + expect(inputProperties(tools.exec)).toEqual(["command", "cwd", "env", "input"]); + }); +}); + +describe("createAITools exec with one backend", () => { + function recordingWorkspace(callable: boolean) { + const calls: Array<{ command: string; backend: string | undefined; input: unknown }> = []; + const workspace = { + runtime: { + async exec( + command: string, + options: { encoding: "utf8"; backend?: string; input?: unknown }, + ) { + calls.push({ command, backend: options.backend, input: options.input }); + return { result: async () => ({ exitCode: 0, stdout: "", stderr: "", value: 1 }) }; + }, + isCallable: (id: string) => callable && id === "worker-javascript", + }, + }; + return { calls, workspace }; + } + + it("has no backend argument and runs on the only backend without defaultBackend", async () => { + const { calls, workspace } = recordingWorkspace(true); + const tools = createAITools({ + workspace, + shell: { backends: { "worker-javascript": { description: "Isolated JavaScript." } } }, + }); + + expect(inputProperties(tools.exec)).toEqual(["command", "cwd", "env", "input"]); + const parsed = inputSchema(tools.exec).parse({ + command: "export default () => 1", + backend: "container", + }); + expect(parsed).toEqual({ command: "export default () => 1" }); + await executeTool(tools.exec, parsed); + expect(calls).toEqual([ + { command: "export default () => 1", backend: "worker-javascript", input: undefined }, + ]); + }); + + it("does not mention backends in the description", () => { + const { workspace } = recordingWorkspace(true); + const tools = createAITools({ + workspace, + shell: { backends: { "worker-javascript": { description: "Isolated JavaScript." } } }, + }); + + expect(toolDescription(tools.exec)).not.toMatch(/backend/i); + }); + + it("drops input for a single shell backend", () => { + const { workspace } = recordingWorkspace(false); + const tools = createAITools({ + workspace, + shell: { backends: { shell: { description: "Fast shell." } } }, + }); + + expect(inputProperties(tools.exec)).toEqual(["command", "cwd", "env"]); + expect(toolDescription(tools.exec)).toContain("Run a shell command"); + expect(toolDescription(tools.exec)).not.toMatch(/backend/i); + }); + + it("keeps the backend argument when more than one backend is configured", () => { + const { workspace } = recordingWorkspace(true); + const tools = createAITools({ + workspace, + shell: { + defaultBackend: "worker-javascript", + backends: { + "worker-javascript": { description: "Isolated JavaScript." }, + container: { description: "Full Linux." }, + }, + }, + }); + + expect(inputProperties(tools.exec)).toEqual(["backend", "command", "cwd", "env", "input"]); + }); + + it("requires defaultBackend when more than one backend is configured", () => { + const { workspace } = recordingWorkspace(false); + + expect(() => + createAITools({ + workspace, + shell: { + backends: { shell: { description: "Fast shell." }, container: { description: "Linux." } }, + }, + }), + ).toThrow(/defaultBackend/); }); }); diff --git a/packages/computer/src/tools/exec.ts b/packages/computer/src/tools/exec.ts index 5a58fb36..620d8e56 100644 --- a/packages/computer/src/tools/exec.ts +++ b/packages/computer/src/tools/exec.ts @@ -3,6 +3,7 @@ import { z } from "zod"; import { notCallableMessage } from "../runtime/runtime.js"; import type { WorkspaceRuntimeValue } from "../runtime/types.js"; +import { truncateText, utf8Prefix } from "../text-truncation.js"; // A finite JSON value: what a callable backend accepts as `input` and // returns as `result`. Declared as a concrete recursive schema rather @@ -64,17 +65,29 @@ export interface ExecWorkspaceLike { // are callable; the runtime derives it from each backend's // `callable` flag. Omit when no backend is callable. isCallable?(id: string): boolean; + // What a backend says about itself for a model, such as the + // language it runs and the modules that code can import. The tool + // shows it after the caller's own description. + describe?(id: string): string | undefined; }; } export interface ExecBackendDescription { - description: string; + // Guidance for the model about this backend, shown before whatever + // the backend says about itself. Required only when the backend does + // not describe itself. + description?: string; } export interface ExecToolOptions { workspace: ExecWorkspaceLike; + // Backends the model may run on. With exactly one, the tool has no + // `backend` argument and always runs there, so the model never has + // to reason about backends. backends: Record; - defaultBackend: string; + // Backend used when the model omits `backend`. Required when more + // than one backend is configured; with one it defaults to that one. + defaultBackend?: string; // Per-snapshot display cap for each of stdout and stderr, in bytes. // Output past it is shown as a truncation marker. Defaults to 64 KiB. maxBytes?: number; @@ -110,16 +123,15 @@ export type ExecToolOutput = } | { command: string; cwd: string | null; backend: string; error: string }; -export function createExecTool(options: ExecToolOptions): Tool< - { - command: string; - cwd?: string; - backend?: string; - env?: Record; - input?: WorkspaceRuntimeValue; - }, - ExecToolOutput -> { +type ExecToolInput = { + command: string; + cwd?: string; + backend?: string; + env?: Record; + input?: WorkspaceRuntimeValue; +}; + +export function createExecTool(options: ExecToolOptions): Tool { const maxBytes = options.maxBytes ?? DEFAULT_MAX_BYTES; const streamMaxBytes = options.streamMaxBytes ?? DEFAULT_STREAM_MAX_BYTES; const now = options.now ?? Date.now; @@ -127,74 +139,66 @@ export function createExecTool(options: ExecToolOptions): Tool< if (backendIds.length === 0) { throw new Error("createExecTool: pass at least one backend in `backends`"); } - if (!backendIds.includes(options.defaultBackend)) { + const single = backendIds.length === 1; + const defaultBackend = options.defaultBackend ?? (single ? backendIds[0] : undefined); + if (defaultBackend === undefined || !backendIds.includes(defaultBackend)) { throw new Error( - `createExecTool: defaultBackend ${JSON.stringify(options.defaultBackend)} is not one of ${backendIds.map((id) => JSON.stringify(id)).join(", ")}`, + `createExecTool: pass a defaultBackend that is one of ${backendIds.map((id) => JSON.stringify(id)).join(", ")}`, ); } - const isCallable = options.workspace.runtime.isCallable?.bind(options.workspace.runtime); - const callableBackendIds = new Set(backendIds.filter((id) => isCallable?.(id) === true)); - const backendGuidance = backendIds - .map((id) => { - const suffix = callableBackendIds.has(id) ? " (callable)" : ""; - return `- ${JSON.stringify(id)}${suffix}: ${options.backends[id].description}`; - }) - .join("\n"); - const callableGuidance = - callableBackendIds.size > 0 - ? [ - "", - `Callable backends (${[...callableBackendIds].map((id) => JSON.stringify(id)).join(", ")}) run \`command\` as module source rather than a shell command. Pass \`input\` to hand the module a structured value, and read the module's returned value back from the \`result\` field. Other backends reject \`input\`.`, - ].join("\n") - : ""; - const description = [ - "Run a shell command in the workspace. The workspace exposes multiple backends, each with different capabilities.", - "Pick the cheapest backend that can run the command; fall back to a heavier one only when the lighter backend's command set doesn't cover what you need.", - "", - "Backends:", - backendGuidance, - "", - `Default backend: ${JSON.stringify(options.defaultBackend)}. Try this first for any command you're not sure about; if it fails with a "command not found" or a similar capability error, retry on a backend whose description covers the missing tool.`, - "Use for builds, test runs, typechecks, formatters, and git plumbing. Prefer the dedicated read, write, and edit tools for file operations. Long output is truncated to keep tool replies small.", - callableGuidance, - ].join("\n"); - - const backendSchema = z - .enum(backendIds as [string, ...string[]]) - .optional() - .describe( - [ - "Which backend to run on. Omit to use the default", - `(${JSON.stringify(options.defaultBackend)}). Set explicitly when the`, - "default backend is not capable of running the command. If a command fails because the backend lacks that tool, retry on a backend whose description covers it.", - ].join(" "), - ); + const runtime = options.workspace.runtime; + const backends = backendIds.map((id) => { + const text = [options.backends[id]?.description, runtime.describe?.(id)] + .filter((part) => part !== undefined && part !== "") + .join("\n\n"); + if (text === "") { + throw new Error( + `createExecTool: backend ${JSON.stringify(id)} does not describe itself; pass a description`, + ); + } + return { id, text, callable: runtime.isCallable?.(id) === true }; + }); + const callableBackendIds = new Set(backends.filter((b) => b.callable).map((b) => b.id)); + const description = describeTool(backends, defaultBackend); + // Offer only the fields that can work: `backend` when there is a + // choice, `input` when some backend accepts it. + const shape: Record = { + command: z.string().describe(commandHint(backends)), + cwd: z.string().optional().describe("Working directory. Defaults to the workspace root."), + env: z + .record(z.string(), z.string()) + .optional() + .describe( + "Environment variables for this run only. Values override the base environment without affecting later runs.", + ), + }; + if (!single) { + shape.backend = z + // SAFETY: createExecTool checked that backendIds has at least one entry. + .enum(backendIds as [string, ...string[]]) + .optional() + .describe( + `Which backend to run on. Omit to use the default (${JSON.stringify(defaultBackend)}). If a command fails because the backend lacks that tool, retry on a backend whose description covers it.`, + ); + } + if (callableBackendIds.size > 0) { + shape.input = jsonValueSchema + .optional() + .describe( + single + ? "Structured value handed to the module." + : "Structured value handed to a callable backend's module. Other backends reject it.", + ); + } + // SAFETY: Every field in `shape` has the type ExecToolInput gives it, and the fields left out are optional there. + const inputSchema = z.object(shape) as unknown as z.ZodType; return tool({ description, - inputSchema: z.object({ - command: z - .string() - .describe( - "Shell command, e.g. 'npm test' or 'git diff HEAD'. For a callable backend this is the module source to run.", - ), - cwd: z.string().optional().describe("Working directory. Defaults to the workspace root."), - backend: backendSchema, - env: z - .record(z.string(), z.string()) - .optional() - .describe( - "Environment variables for this run only. Values override the backend's base environment without affecting later runs.", - ), - input: jsonValueSchema - .optional() - .describe( - "Structured value handed to a callable backend's module. Only callable backends accept it; other backends reject it.", - ), - }), + inputSchema, execute: async function* ({ command, cwd, backend, env, input }, { abortSignal }) { - const selectedBackend = backend ?? options.defaultBackend; + const selectedBackend = backend ?? defaultBackend; const base = { command, cwd: cwd ?? null, backend: selectedBackend }; if (input !== undefined && !callableBackendIds.has(selectedBackend)) { yield { ...base, error: notCallableMessage(selectedBackend) }; @@ -285,8 +289,8 @@ export function createExecTool(options: ExecToolOptions): Tool< yield { ...base, exitCode: result.exitCode, - stdout: truncate(result.stdout, maxBytes), - stderr: truncate(result.stderr, maxBytes), + stdout: truncateText(result.stdout, maxBytes), + stderr: truncateText(result.stderr, maxBytes), ...(result.value === undefined ? {} : { result: result.value }), }; } catch (err) { @@ -297,6 +301,62 @@ export function createExecTool(options: ExecToolOptions): Tool< }); } +const FILE_TOOLS_HINT = + "Prefer the dedicated read, write, and edit tools for file operations. Long output is truncated to keep tool replies small."; +const SHELL_HINT = "Use for builds, test runs, typechecks, formatters, and git plumbing."; +const CALLABLE_HINT = + "Pass `input` to hand the module a structured value, and read its return value back from the `result` field."; + +interface DescribedBackend { + readonly id: string; + readonly text: string; + readonly callable: boolean; +} + +// With one backend the description is about what it does. With several +// it lists them and explains how to choose. +function describeTool(backends: readonly DescribedBackend[], defaultBackend: string): string { + const [only, ...others] = backends; + if (only !== undefined && others.length === 0) { + return only.callable + ? ["Run code in the workspace.", "", only.text, "", CALLABLE_HINT, FILE_TOOLS_HINT].join("\n") + : [ + "Run a shell command in the workspace.", + "", + only.text, + "", + `${SHELL_HINT} ${FILE_TOOLS_HINT}`, + ].join("\n"); + } + const callable = backends.filter((backend) => backend.callable).map((b) => JSON.stringify(b.id)); + return [ + "Run a shell command in the workspace. The workspace exposes multiple backends, each with different capabilities.", + "Pick the cheapest backend that can run the command; fall back to a heavier one only when the lighter backend's command set doesn't cover what you need.", + "", + "Backends:", + ...backends.map( + (b) => `- ${JSON.stringify(b.id)}${b.callable ? " (callable)" : ""}: ${b.text}`, + ), + "", + `Default backend: ${JSON.stringify(defaultBackend)}. Try this first for any command you're not sure about; if it fails with a "command not found" or a similar capability error, retry on a backend whose description covers the missing tool.`, + `${SHELL_HINT} ${FILE_TOOLS_HINT}`, + ...(callable.length === 0 + ? [] + : [ + "", + `Callable backends (${callable.join(", ")}) run \`command\` as module source rather than a shell command. ${CALLABLE_HINT} Other backends reject \`input\`.`, + ]), + ].join("\n"); +} + +function commandHint(backends: readonly DescribedBackend[]): string { + if (backends.every((backend) => backend.callable)) return "Module source to run."; + if (backends.every((backend) => !backend.callable)) { + return "Shell command, e.g. 'npm test' or 'git diff HEAD'."; + } + return "Shell command, e.g. 'npm test' or 'git diff HEAD'. For a callable backend this is the module source to run."; +} + function errorMessage(err: unknown): string { return err instanceof Error ? err.message : String(err); } @@ -329,47 +389,16 @@ class StreamBuffer { } // The chunk crosses the cap: keep the largest whole-character // prefix that fits, then stop growing the head. - let used = this.#headBytes; - let end = 0; - for (const char of chunk) { - const charBytes = encoder.encode(char).byteLength; - if (used + charBytes > this.#cap) break; - used += charBytes; - end += char.length; - } - this.#head += chunk.slice(0, end); - this.#headBytes = used; + const prefix = utf8Prefix(chunk, this.#cap - this.#headBytes); + this.#head += prefix.text; + this.#headBytes += prefix.bytes; } render(maxBytes: number): string { if (this.#totalBytes <= maxBytes && this.#totalBytes === this.#headBytes) { return this.#head; } - let used = 0; - let end = 0; - for (const char of this.#head) { - const charBytes = encoder.encode(char).byteLength; - if (used + charBytes > maxBytes) break; - used += charBytes; - end += char.length; - } - return `${this.#head.slice(0, end)}\n\n[truncated, ${this.#totalBytes - used} more bytes]`; + const shown = utf8Prefix(this.#head, maxBytes); + return `${shown.text}\n\n[truncated, ${this.#totalBytes - shown.bytes} more bytes]`; } } - -function truncate(value: string, maxBytes: number): string { - if (!value) return value; - const totalBytes = encoder.encode(value).byteLength; - if (totalBytes <= maxBytes) return value; - - let usedBytes = 0; - let endOffset = 0; - for (const char of value) { - const charBytes = encoder.encode(char).byteLength; - if (usedBytes + charBytes > maxBytes) break; - usedBytes += charBytes; - endOffset += char.length; - } - - return `${value.slice(0, endOffset)}\n\n[truncated, ${totalBytes - usedBytes} more bytes]`; -}