diff --git a/.changeset/README.md b/.changeset/README.md new file mode 100644 index 0000000000..e5b6d8d6a6 --- /dev/null +++ b/.changeset/README.md @@ -0,0 +1,8 @@ +# Changesets + +Hello and welcome! This folder has been automatically generated by `@changesets/cli`, a build tool that works +with multi-package repos, or single-package repos to help you version and publish your code. You can +find the full documentation for it [in our repository](https://github.com/changesets/changesets) + +We have a quick list of common questions to get you started engaging with this project in +[our documentation](https://github.com/changesets/changesets/blob/main/docs/common-questions.md) diff --git a/.changeset/config.json b/.changeset/config.json new file mode 100644 index 0000000000..4b1f0b3b5f --- /dev/null +++ b/.changeset/config.json @@ -0,0 +1,17 @@ +{ + "$schema": "https://unpkg.com/@changesets/config@3.1.2/schema.json", + "changelog": [ + "@changesets/changelog-github", + { "repo": "cloudflare/agents" } + ], + "commit": false, + "fixed": [], + "linked": [], + "access": "public", + "baseBranch": "main", + "updateInternalDependencies": "minor", + "ignore": ["@cloudflare/agents-*"], + "___experimentalUnsafeOptions_WILL_CHANGE_IN_PATCH": { + "onlyUpdatePeerDependentsWhenOutOfRange": true + } +} diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md new file mode 100644 index 0000000000..90838b59cf --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -0,0 +1,30 @@ +--- +name: Bug report +about: Create a report to help us improve Agents SDK +title: "" +labels: "" +assignees: "" +--- + +**Describe the bug** +A clear and concise description of what the bug is. + +**To Reproduce** +Steps to reproduce the behavior: + +1. Go to '...' +2. Click on '....' +3. Scroll down to '....' +4. See error + +**Expected behavior** +A clear and concise description of what you expected to happen. + +**Screenshots** +If applicable, add screenshots to help explain your problem. + +**Version:** +What version of Agents SDK are you having issues with. + +**Additional context** +Add any other context about the problem here. diff --git a/.github/changeset-publish.ts b/.github/changeset-publish.ts new file mode 100644 index 0000000000..ed97a9536b --- /dev/null +++ b/.github/changeset-publish.ts @@ -0,0 +1,8 @@ +import { execSync } from "node:child_process"; + +execSync("npx tsx ./.github/resolve-workspace-versions.ts", { + stdio: "inherit" +}); +execSync("npx changeset publish", { + stdio: "inherit" +}); diff --git a/.github/changeset-version.ts b/.github/changeset-version.ts new file mode 100644 index 0000000000..eabfcb2873 --- /dev/null +++ b/.github/changeset-version.ts @@ -0,0 +1,13 @@ +import { execSync } from "node:child_process"; + +// This script is used by the `release.yml` workflow to update the version of the packages being released. +// The standard step is only to run `changeset version` but this does not update the package-lock.json file. +// So we also run `npm install`, which does this update. +// This is a workaround until this is handled automatically by `changeset version`. +// See https://github.com/changesets/changesets/issues/421. +execSync("npx changeset version", { + stdio: "inherit" +}); +execSync("npm install", { + stdio: "inherit" +}); diff --git a/.github/resolve-workspace-versions.ts b/.github/resolve-workspace-versions.ts new file mode 100644 index 0000000000..c55caa0b83 --- /dev/null +++ b/.github/resolve-workspace-versions.ts @@ -0,0 +1,89 @@ +// this looks for all package.jsons in /packages/**/package.json +// and replaces it with the actual version ids + +import * as fs from "node:fs"; +import fg from "fast-glob"; + +// we do this in 2 passes +// first let's cycle through all packages and get thier version numbers + +/** + * Minimal interface for the subset of package.json fields this script reads + * and writes. The index signature allows dynamic access to dependency fields + * while the explicit properties give type-safe access to name/version. + */ +interface PackageJson { + name: string; + version: string; + dependencies?: Record; + devDependencies?: Record; + peerDependencies?: Record; + optionalDependencies?: Record; + [key: string]: unknown; +} + +const packageJsons: Record = + {}; + +for await (const file of await fg.glob( + "./(packages|examples|guides)/*/package.json" +)) { + let packageJson: PackageJson; + try { + packageJson = JSON.parse(fs.readFileSync(file, "utf8")) as PackageJson; + } catch (err) { + console.error(`Failed to parse ${file}:`, err); + continue; + } + packageJsons[packageJson.name] = { + file, + packageJson + }; +} + +// then we'll revisit them, and replace any "workspace:*" references +// with "^(actual version)" + +for (const [packageName, { file, packageJson }] of Object.entries( + packageJsons +)) { + let changed = false; + const depFields = [ + "dependencies", + "devDependencies", + "peerDependencies", + "optionalDependencies" + ] as const; + for (const field of depFields) { + const deps = packageJson[field]; + if (!deps) continue; + for (const [dependencyName, currentRange] of Object.entries(deps)) { + if (dependencyName in packageJsons) { + // For peerDependencies, preserve intentionally wide ranges. + // Only rewrite workspace:* references or exact caret ranges + // that match the previous version (i.e. changesets-managed). + if ( + field === "peerDependencies" && + !currentRange.startsWith("workspace:") && + !currentRange.startsWith("^") + ) { + continue; + } + + let actualVersion = packageJsons[dependencyName].packageJson.version; + if (!actualVersion.startsWith("0.0.0-")) { + actualVersion = `^${actualVersion}`; + } + + console.log( + `${packageName}: setting ${field}.${dependencyName} to ${actualVersion}` + ); + deps[dependencyName] = actualVersion; + changed = true; + } + } + } + if (changed) { + fs.writeFileSync(file, `${JSON.stringify(packageJson, null, 2)}\n`); + } +} diff --git a/.github/version-script.ts b/.github/version-script.ts new file mode 100644 index 0000000000..e269322d22 --- /dev/null +++ b/.github/version-script.ts @@ -0,0 +1,27 @@ +import * as fs from "node:fs"; +import { execSync } from "node:child_process"; +async function main() { + try { + console.log("Getting current git hash..."); + const stdout = execSync("git rev-parse --short HEAD").toString(); + console.log("Git hash:", stdout.trim()); + + for (const path of [ + "./packages/agents/package.json", + "./packages/hono-agents/package.json" + ]) { + const packageJson = JSON.parse(fs.readFileSync(path, "utf-8")); + packageJson.version = `0.0.0-${stdout.trim()}`; + fs.writeFileSync(path, `${JSON.stringify(packageJson, null, 2)}\n`); + } + } catch (error) { + console.error(error); + process.exit(1); + } +} + +main().catch((err) => { + // Build failures should fail + console.error(err); + process.exit(1); +}); diff --git a/.github/workflows/bonk.yml b/.github/workflows/bonk.yml new file mode 100644 index 0000000000..8f84b6c6bc --- /dev/null +++ b/.github/workflows/bonk.yml @@ -0,0 +1,34 @@ +name: Bonk + +on: + issue_comment: + types: [created] + pull_request_review_comment: + types: [created] + +jobs: + bonk: + if: github.event.sender.type != 'Bot' + runs-on: ubuntu-latest + timeout-minutes: 20 + concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: false + permissions: + id-token: write + issues: write + pull-requests: write + contents: write + steps: + - name: Checkout repository + uses: actions/checkout@v6 + + - name: Run Bonk + uses: ask-bonk/ask-bonk/github@main + env: + CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CF_AI_GATEWAY_ACCOUNT_ID }} + CLOUDFLARE_GATEWAY_ID: ${{ secrets.CF_AI_GATEWAY_NAME }} + CLOUDFLARE_API_TOKEN: ${{ secrets.CF_AI_GATEWAY_TOKEN }} + with: + model: "cloudflare-ai-gateway/anthropic/claude-opus-4-6" + mentions: "/bonk,@ask-bonk" diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml new file mode 100644 index 0000000000..aadfb766ab --- /dev/null +++ b/.github/workflows/nightly.yml @@ -0,0 +1,76 @@ +name: Nightly E2E + +on: + schedule: + # Every day at midnight UTC + - cron: "0 0 * * *" + # Allow manual trigger for debugging + workflow_dispatch: + +concurrency: + group: ${{ github.workflow }} + cancel-in-progress: true + +jobs: + e2e-ai-chat: + name: "E2E: ai-chat (Playwright)" + timeout-minutes: 15 + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v6 + with: + fetch-depth: 1 + + - uses: actions/setup-node@v6 + with: + node-version: 24 + cache: "npm" + + - run: npm ci + - run: npm run build + + - name: Get Playwright version + id: playwright-version + run: echo "version=$(jq -r '.packages["node_modules/playwright"].version' package-lock.json)" >> $GITHUB_OUTPUT + + - name: Cache Playwright browsers + uses: actions/cache@v5 + id: playwright-cache + with: + path: ~/.cache/ms-playwright + key: ${{ runner.os }}-playwright-${{ steps.playwright-version.outputs.version }} + + - name: Install Playwright browsers + if: steps.playwright-cache.outputs.cache-hit != 'true' + run: npx playwright install --with-deps chromium + + - name: Run ai-chat e2e tests + env: + CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }} + CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} + run: npx playwright test --config e2e/playwright.config.ts + working-directory: packages/ai-chat + + e2e-think: + name: "E2E: think (chat recovery)" + timeout-minutes: 10 + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v6 + with: + fetch-depth: 1 + + - uses: actions/setup-node@v6 + with: + node-version: 24 + cache: "npm" + + - run: npm ci + - run: npm run build + + - name: Run think e2e tests + env: + CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }} + CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} + run: npm run test:e2e + working-directory: packages/think diff --git a/.github/workflows/pullrequest.yml b/.github/workflows/pullrequest.yml new file mode 100644 index 0000000000..de95890fc4 --- /dev/null +++ b/.github/workflows/pullrequest.yml @@ -0,0 +1,61 @@ +name: Pull Request + +on: + pull_request: + paths-ignore: + - "docs/**" + - "design/**" + - "*.md" + - ".changeset/**" + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number }} + cancel-in-progress: true + +env: + PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: "1" + +jobs: + ci: + timeout-minutes: 20 + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v6 + with: + fetch-depth: 1 + + - uses: actions/setup-node@v6 + with: + node-version: 24 + cache: "npm" + + - name: Cache Nx + uses: actions/cache@v5 + with: + path: .nx/cache + key: ${{ runner.os }}-nx-${{ hashFiles('package-lock.json') }}-${{ github.sha }} + restore-keys: | + ${{ runner.os }}-nx-${{ hashFiles('package-lock.json') }}- + ${{ runner.os }}-nx- + + - run: npm ci + - run: npm run build + - run: npm run check + + - name: Get Playwright version + id: playwright-version + run: echo "version=$(jq -r '.packages["node_modules/playwright"].version' package-lock.json)" >> $GITHUB_OUTPUT + + - name: Cache Playwright browsers + uses: actions/cache@v5 + id: playwright-cache + with: + path: ~/.cache/ms-playwright + key: ${{ runner.os }}-playwright-${{ steps.playwright-version.outputs.version }} + + - name: Install Playwright browsers + if: steps.playwright-cache.outputs.cache-hit != 'true' + run: npx playwright install --with-deps chromium + + - run: CI=true npx nx run-many -t test + - run: npx pkg-pr-new publish --peerDeps ./packages/* diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000000..0ddc6d64a3 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,47 @@ +name: Release + +on: + push: + branches: + - main + +concurrency: + group: ${{ github.workflow }} + cancel-in-progress: false + +env: + PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: "1" + +jobs: + release: + permissions: + id-token: write # Required for trusted publishing + contents: write + pull-requests: write + + if: ${{ github.repository_owner == 'cloudflare' }} + timeout-minutes: 5 + runs-on: ubuntu-24.04 + + steps: + - uses: actions/checkout@v6 + with: + fetch-depth: 1 + + - uses: actions/setup-node@v6 + with: + node-version: 24 + registry-url: "https://registry.npmjs.org" + cache: "npm" + + - run: npm ci + - run: npm run build + + - id: changesets + uses: changesets/action@v1.7.0 + with: + version: npx tsx .github/changeset-version.ts + publish: npx tsx .github/changeset-publish.ts + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + NPM_CONFIG_PROVENANCE: true diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000000..4c89e7e37c --- /dev/null +++ b/.gitignore @@ -0,0 +1,155 @@ +# Logs +logs +*.log +npm-debug.log* +yarn-debug.log* +yarn-error.log* +lerna-debug.log* +.pnpm-debug.log* + +# Diagnostic reports (https://nodejs.org/api/report.html) +report.[0-9]*.[0-9]*.[0-9]*.[0-9]*.json + +# Runtime data +pids +*.pid +*.seed +*.pid.lock + +# Directory for instrumented libs generated by jscoverage/JSCover +lib-cov + +# Coverage directory used by tools like istanbul +coverage +*.lcov + +# nyc test coverage +.nyc_output + +# Grunt intermediate storage (https://gruntjs.com/creating-plugins#storing-task-files) +.grunt + +# Bower dependency directory (https://bower.io/) +bower_components + +# node-waf configuration +.lock-wscript + +# Compiled binary addons (https://nodejs.org/api/addons.html) +build/Release + +# Dependency directories +node_modules/ +jspm_packages/ + +# Snowpack dependency directory (https://snowpack.dev/) +web_modules/ + +# TypeScript cache +*.tsbuildinfo + +# Optional npm cache directory +.npm + +# Optional eslint cache +.eslintcache + +# Optional stylelint cache +.stylelintcache + +# Microbundle cache +.rpt2_cache/ +.rts2_cache_cjs/ +.rts2_cache_es/ +.rts2_cache_umd/ + +# Optional REPL history +.node_repl_history + +# Output of 'npm pack' +*.tgz + +# Yarn Integrity file +.yarn-integrity + +# dotenv environment variable files +.env +.env.development.local +.env.test.local +.env.production.local +.env.local + +# parcel-bundler cache (https://parceljs.org/) +.cache +.parcel-cache + +# Next.js build output +.next +out + +# Nuxt.js build / generate output +.nuxt +dist + +# Gatsby files +.cache/ +# Comment in the public line in if your project uses Gatsby and not Next.js +# https://nextjs.org/blog/next-9-1#public-directory-support +# public + +# vuepress build output +.vuepress/dist + +# vuepress v2.x temp and cache directory +.temp +.cache + +# vitepress build output +**/.vitepress/dist + +# vitepress cache directory +**/.vitepress/cache + +# Docusaurus cache and generated files +.docusaurus + +# Serverless directories +.serverless/ + +# FuseBox cache +.fusebox/ + +# DynamoDB Local files +.dynamodb/ + +# TernJS port file +.tern-port + +# Stores VSCode versions used for testing VSCode extensions +.vscode-test + +# yarn v2 +.yarn/cache +.yarn/unplugged +.yarn/build-state.yml +.yarn/install-state.gz +.pnp.* + +# cloudflare/wrangler +.wrangler +.wrangler-*-state +.dev.vars + +# macOS +.DS_Store + +.astro +# Vitest browser test screenshots +__screenshots__ + +# Playwright test artifacts +test-results/ + +# Nx +.nx/cache +.nx/workspace-data diff --git a/.husky/pre-commit b/.husky/pre-commit new file mode 100644 index 0000000000..2312dc587f --- /dev/null +++ b/.husky/pre-commit @@ -0,0 +1 @@ +npx lint-staged diff --git a/.mcp.json b/.mcp.json new file mode 100644 index 0000000000..654664a7d6 --- /dev/null +++ b/.mcp.json @@ -0,0 +1,8 @@ +{ + "mcpServers": { + "agents": { + "type": "http", + "url": "https://agents.cloudflare.com/mcp" + } + } +} diff --git a/.nxignore b/.nxignore new file mode 100644 index 0000000000..187847bd9b --- /dev/null +++ b/.nxignore @@ -0,0 +1,4 @@ +node_modules +dist +.wrangler +.astro diff --git a/.oxfmtrc.json b/.oxfmtrc.json new file mode 100644 index 0000000000..67adf235cc --- /dev/null +++ b/.oxfmtrc.json @@ -0,0 +1,7 @@ +{ + "$schema": "./node_modules/oxfmt/configuration_schema.json", + "trailingComma": "none", + "printWidth": 80, + "experimentalSortPackageJson": false, + "ignorePatterns": ["packages/agents/CHANGELOG.md", "site/agents/.astro"] +} diff --git a/.oxlintrc.json b/.oxlintrc.json new file mode 100644 index 0000000000..20062223ff --- /dev/null +++ b/.oxlintrc.json @@ -0,0 +1,23 @@ +{ + "$schema": "./node_modules/oxlint/configuration_schema.json", + "plugins": ["react", "jsx-a11y", "typescript"], + "categories": { + "correctness": "error" + }, + "rules": { + "no-explicit-any": "error", + "no-unused-expressions": "off", + "typescript/no-deprecated": "warn", + "no-this-alias": "off", + "react-hooks/exhaustive-deps": "warn", + "no-unused-vars": [ + "error", + { + "argsIgnorePattern": "^_", + "varsIgnorePattern": "^_", + "caughtErrorsIgnorePattern": "^_" + } + ] + }, + "ignorePatterns": ["**/env.d.ts"] +} diff --git a/.vscode/extensions.json b/.vscode/extensions.json new file mode 100644 index 0000000000..99e2f7ddf7 --- /dev/null +++ b/.vscode/extensions.json @@ -0,0 +1,3 @@ +{ + "recommendations": ["oxc.oxc-vscode"] +} diff --git a/.vscode/settings.json b/.vscode/settings.json new file mode 100644 index 0000000000..fe10f2fd84 --- /dev/null +++ b/.vscode/settings.json @@ -0,0 +1,25 @@ +{ + "editor.defaultFormatter": "oxc.oxc-vscode", + "editor.formatOnSave": true, + "editor.formatOnPaste": false, + "editor.formatOnType": false, + "oxc.fmt.experimental": true, + "[javascript]": { + "editor.defaultFormatter": "oxc.oxc-vscode" + }, + "[typescript]": { + "editor.defaultFormatter": "oxc.oxc-vscode" + }, + "[javascriptreact]": { + "editor.defaultFormatter": "oxc.oxc-vscode" + }, + "[typescriptreact]": { + "editor.defaultFormatter": "oxc.oxc-vscode" + }, + "[json]": { + "editor.defaultFormatter": "oxc.oxc-vscode" + }, + "[jsonc]": { + "editor.defaultFormatter": "oxc.oxc-vscode" + } +} diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000000..a9fbde8202 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,192 @@ +# AGENTS.md + +## Project overview + +Cloudflare Agents SDK — a framework for building stateful AI agents on Cloudflare Workers. This is a monorepo containing the core SDK packages, examples, guides, sites, and documentation. + +## Repository structure + +``` +packages/ # Published npm packages (need changesets for changes) + agents/ # Core SDK (see packages/agents/AGENTS.md) + ai-chat/ # @cloudflare/ai-chat — higher-level AI chat agent + hono-agents/ # Hono framework integration + codemode/ # @cloudflare/codemode — experimental code generation + +examples/ # Self-contained demo apps (see examples/AGENTS.md) + playground/ # Main showcase app — all SDK features in one UI (uses Kumo design system) + mcp/ # MCP server example + mcp-client/ # MCP client example + ... # ~20 examples total + +experimental/ # Work-in-progress experiments (not published, no stability guarantees) + +site/ # Deployed websites + agents/ # agents.cloudflare.com (Astro) + ai-playground/ # Workers AI playground (React + Vite) + +guides/ # In-depth pattern tutorials with narrative READMEs (see guides/AGENTS.md) + anthropic-patterns/ + human-in-the-loop/ + +openai-sdk/ # Examples using @openai/agents SDK + basic/ chess-app/ handoffs/ human-in-the-loop/ ... + +docs/ # Markdown docs for developers.cloudflare.com (see docs/AGENTS.md) +design/ # Architecture and design decision records (see design/AGENTS.md) +scripts/ # Repo-wide tooling (typecheck, export checks, update checks) +``` + +## Nested AGENTS.md files + +Some directories have their own AGENTS.md with deeper guidance: + +| File | Scope | +| --------------------------- | ------------------------------------------------------------------------- | +| `packages/agents/AGENTS.md` | Core SDK internals — exports, source layout, build, testing, architecture | +| `examples/AGENTS.md` | Example conventions — required structure, consistency rules, known issues | +| `guides/AGENTS.md` | Guide conventions — how guides differ from examples, README expectations | +| `docs/AGENTS.md` | Writing user-facing docs — Diátaxis framework, upstream sync, style | +| `design/AGENTS.md` | Design records and RFCs — format, workflow, relationship to docs | + +## Setup + +```bash +npm install # installs all workspaces; postinstall runs patch-package + playwright +``` + +Node 24+ required. Uses npm workspaces with [Nx](https://nx.dev) for task orchestration, caching, and affected detection. + +## Commands + +Run from the repo root: + +| Command | What it does | +| ---------------------------- | ------------------------------------------------------------------ | +| `npm run build` | Builds all packages via Nx (cached, dependency-ordered) | +| `npm run check` | Full CI check: sherif + export checks + oxfmt + oxlint + typecheck | +| `npm run test` | Runs all tests via Nx (cached) | +| `npm run test:react` | Runs Playwright-based React hook tests for agents | +| `npm run typecheck` | TypeScript type checking across the repo (custom script) | +| `npm run format` | Oxfmt format all files | +| `npm run check:exports` | Verifies package.json exports match actual build output | +| `npx nx affected -t build` | Build only packages affected by current changes | +| `npx nx affected -t test` | Test only packages affected by current changes | +| `npx nx run :build` | Build a single project (and its dependencies) | + +Run an example locally: + +```bash +cd examples/playground # or any example +npm run dev # starts Vite dev server + Workers runtime via @cloudflare/vite-plugin +``` + +## Code standards + +### TypeScript + +- Strict mode enabled (`agents/tsconfig`) +- Target: ES2021, module: ES2022, moduleResolution: bundler +- `verbatimModuleSyntax: true` — use explicit `import type` for type-only imports +- JSX: `react-jsx` + +### Linting — Oxlint + +Config in `.oxlintrc.json`. Plugins: `react`, `jsx-a11y`, `typescript`, `react-hooks`. Key rules: + +- `no-explicit-any: "error"` — never use `any`, use `unknown` and narrow +- `no-unused-vars: "error"` — with `varsIgnorePattern: "^_"` and `argsIgnorePattern: "^_"` +- `correctness` category set to `"error"` — catches common mistakes +- `jsx-a11y` rules enabled — accessibility violations are errors +- `react-hooks/exhaustive-deps: "warn"` — warns on missing hook dependencies + +Oxlint does **not** handle formatting — Oxfmt does. + +### Formatting — Oxfmt + +- Run `npm run format` or rely on lint-staged (auto-formats on commit via husky) +- Config in `.oxfmtrc.json` (`trailingComma: "none"`, `printWidth: 80`) + +### Workers conventions + +- Always TypeScript, always ES modules +- `wrangler.jsonc` (not `.toml`) for configuration +- All wrangler configs use `compatibility_date: "2026-01-28"` and `compatibility_flags: ["nodejs_compat"]` +- Never hardcode secrets — use `wrangler secret put` or `.dev.vars` +- No native/FFI dependencies (must run in Workers runtime) + +## Testing + +Tests use **vitest** with `@cloudflare/vitest-pool-workers` for running inside the Workers runtime. + +```bash +npm run test # agents + ai-chat unit/integration tests +npm run test:react # Playwright-based React hook tests (agents package) +``` + +Test locations: + +- `packages/agents/src/tests/` — core SDK tests +- `packages/agents/src/react-tests/` — React hook tests (Playwright + vitest-browser-react) +- `packages/ai-chat/src/tests/` — AI chat tests +- `packages/agents/src/tests-d/` — type-level tests (`.test-d.ts`) + +Each test directory has its own `vitest.config.ts` and (for Workers tests) a `wrangler.jsonc`. + +## Contributing + +### Changesets + +Changes to `packages/` that affect the public API or fix bugs need a changeset: + +```bash +npx changeset # interactive prompt — pick packages, semver bump, description +``` + +This creates a markdown file in `.changeset/` that gets consumed during release. + +Examples, guides, and sites don't need changesets. + +### Pull request process + +CI runs on every PR (`npm ci && npm run build && npm run check && npm run test`). All checks must pass. The workflow is in `.github/workflows/pullrequest.yml`. + +### Generated files + +- `env.d.ts` files are generated by `wrangler types` — regenerate with `npx wrangler types` inside the relevant example/package, don't hand-edit +- `package-lock.json` — regenerated by `npm install`, don't hand-edit + +## Learned Workspace Facts + +- `packages/shell/` is published as `@cloudflare/shell` — an experimental sandboxed JS execution and filesystem runtime for agents, built on the same dynamic Worker loader machinery as `@cloudflare/codemode`. +- To run code against a `Workspace`: import `stateTools` from `@cloudflare/shell/workers` and `DynamicWorkerExecutor`/`resolveProvider` from `@cloudflare/codemode`; use `executor.execute(code, [resolveProvider(stateTools(workspace))])`. + +## Learned User Preferences + +- Keep `Workspace` as a pure durable filesystem — do not embed execution or session logic inside it. Execution is a caller concern wired via `@cloudflare/codemode` + `stateTools`. +- When a package boundary feels wrong (e.g., a helper package depending on a larger package just for an adapter), prefer moving the adapter out rather than carrying the dependency. + +## Boundaries + +**Always:** + +- Run `npm run check` before considering work done +- Use `import type` for type-only imports (enforced by `verbatimModuleSyntax`) +- Keep examples simple and self-contained — they're user-facing learning material +- Use Cloudflare Workers APIs (KV, D1, R2, Durable Objects, etc.) over third-party equivalents +- Use Workers AI for LLM calls in examples — not third-party APIs like OpenAI or Anthropic + +**Ask first:** + +- Adding new dependencies to `packages/` (these ship to users) +- Changing `wrangler.jsonc` compatibility dates across the repo +- Modifying CI workflows + +**Never:** + +- Hardcode secrets or API keys +- Add native/FFI/C-binding dependencies +- Use `any` — Oxlint will reject it +- Use CommonJS or Service Worker format — ES modules only +- Modify `node_modules/` or `dist/` directories +- Force push to main diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000000..61cc9d4625 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License Copyright (c) 2025 Cloudflare, Inc. + +Permission is hereby granted, free of +charge, to any person obtaining a copy of this software and associated +documentation files (the "Software"), to deal in the Software without +restriction, including without limitation the rights to use, copy, modify, merge, +publish, distribute, sublicense, and/or sell copies of the Software, and to +permit persons to whom the Software is furnished to do so, subject to the +following conditions: + +The above copyright notice and this permission notice +(including the next paragraph) shall be included in all copies or substantial +portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF +ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO +EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR +OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN +THE SOFTWARE. diff --git a/NOTICE b/NOTICE new file mode 100644 index 0000000000..a351ce584f --- /dev/null +++ b/NOTICE @@ -0,0 +1,27 @@ +NOTICE + +This project incorporates code and inspiration from the following open source projects: + +1. Anthropic MCP TS SDK + - License: MIT License + - Repository: https://github.com/modelcontextprotocol/typescript-sdk + - Copyright: Anthropic, PBC + +2. Coinbase x402 + - License: Apache License 2.0 + - Repository: https://github.com/coinbase/x402 + - Copyright: Coinbase, Inc. + +3. ethanniser/x402-mcp + - License: MIT License + - Repository: https://github.com/ethanniser/x402-mcp + - Copyright: @ethanniser + +4. Vercel AI SDK + - License: Apache License 2.0 + - Repository: https://github.com/vercel/ai + - Copyright: Vercel, Inc. + +The original licenses and copyright notices from these projects are preserved and should be included with any distribution of this software. + +For a complete list of third-party licenses, see [THIRD_PARTY_LICENSES.md](THIRD_PARTY_LICENSES.md). diff --git a/README.md b/README.md new file mode 100644 index 0000000000..7b5f356057 --- /dev/null +++ b/README.md @@ -0,0 +1,178 @@ +# Cloudflare Agents + +[![npm version](https://img.shields.io/npm/v/agents)](https://www.npmjs.com/package/agents) +[![npm downloads](https://img.shields.io/npm/dw/agents)](https://www.npmjs.com/package/agents) + +![npm install agents](assets/npm-install-agents.svg) + +Agents are persistent, stateful execution environments for agentic workloads, powered by Cloudflare [Durable Objects](https://developers.cloudflare.com/durable-objects/). Each agent has its own state, storage, and lifecycle — with built-in support for real-time communication, scheduling, AI model calls, MCP, workflows, and more. + +Agents hibernate when idle and wake on demand. You can run millions of them — one per user, per session, per game room — each costs nothing when inactive. + +```sh +npm create cloudflare@latest -- --template cloudflare/agents-starter +``` + +Or add to an existing project: + +```sh +npm install agents +``` + +**[Read the docs](https://developers.cloudflare.com/agents/)** — getting started, API reference, guides, and more. + +## Quick Example + +A counter agent with persistent state, callable methods, and real-time sync to a React frontend: + +```typescript +// server.ts +import { Agent, routeAgentRequest, callable } from "agents"; + +export type CounterState = { count: number }; + +export class CounterAgent extends Agent { + initialState = { count: 0 }; + + @callable() + increment() { + this.setState({ count: this.state.count + 1 }); + return this.state.count; + } + + @callable() + decrement() { + this.setState({ count: this.state.count - 1 }); + return this.state.count; + } +} + +export default { + async fetch(request: Request, env: Env, ctx: ExecutionContext) { + return ( + (await routeAgentRequest(request, env)) ?? + new Response("Not found", { status: 404 }) + ); + } +}; +``` + +```tsx +// client.tsx +import { useAgent } from "agents/react"; +import { useState } from "react"; +import type { CounterAgent, CounterState } from "./server"; + +function Counter() { + const [count, setCount] = useState(0); + + const agent = useAgent({ + agent: "CounterAgent", + onStateUpdate: (state) => setCount(state.count) + }); + + return ( +
+ {count} + + +
+ ); +} +``` + +State changes sync to all connected clients automatically. Call methods like they're local functions. + +## Features + +| Feature | Description | +| --------------------- | ---------------------------------------------------------------------- | +| **Persistent State** | Syncs to all connected clients, survives restarts | +| **Callable Methods** | Type-safe RPC via the `@callable()` decorator | +| **Scheduling** | One-time, recurring, and cron-based tasks | +| **WebSockets** | Real-time bidirectional communication with lifecycle hooks | +| **AI Chat** | Message persistence, resumable streaming, server/client tool execution | +| **MCP** | Act as MCP servers or connect as MCP clients | +| **Workflows** | Durable multi-step tasks with human-in-the-loop approval | +| **Email** | Receive and respond via Cloudflare Email Routing | +| **Code Mode** | LLMs generate executable TypeScript instead of individual tool calls | +| **SQL** | Direct SQLite queries via Durable Objects | +| **React Hooks** | `useAgent` and `useAgentChat` for frontend integration | +| **Vanilla JS Client** | `AgentClient` for non-React environments | + +**Coming soon:** Realtime voice agents, web browsing (headless browser), sandboxed code execution, and multi-channel communication (SMS, messengers). + +## Packages + +| Package | Description | +| ------------------------------------------- | ------------------------------------------------------------------------------- | +| [`agents`](packages/agents) | Core SDK — Agent class, routing, state, scheduling, MCP, email, workflows | +| [`@cloudflare/ai-chat`](packages/ai-chat) | Higher-level AI chat — persistent messages, resumable streaming, tool execution | +| [`hono-agents`](packages/hono-agents) | Hono middleware for adding agents to Hono apps | +| [`@cloudflare/codemode`](packages/codemode) | Experimental — LLMs write executable code to orchestrate tools | + +## Examples + +The [`examples/`](examples) directory has self-contained demos covering most SDK features — MCP servers/clients, workflows, email agents, webhooks, tic-tac-toe, resumable streaming, and more. The [`playground`](examples/playground) is the kitchen-sink showcase with everything in one UI. + +There are also examples using the [OpenAI Agents SDK](https://openai.github.io/openai-agents-js/) in [`openai-sdk/`](openai-sdk). + +Run any example locally: + +```sh +cd examples/playground +npm run dev +``` + +## Documentation + +- [Full docs](https://developers.cloudflare.com/agents/) on developers.cloudflare.com +- [`docs/`](docs) directory in this repo (synced upstream) +- [Anthropic Patterns guide](guides/anthropic-patterns) — sequential, routing, parallel, orchestrator, evaluator +- [Human-in-the-Loop guide](guides/human-in-the-loop) — approval workflows with pause/resume + +## Repository Structure + +| Directory | Description | +| ----------------------------------------------- | -------------------------------------------------------- | +| [`packages/agents/`](packages/agents) | Core SDK | +| [`packages/ai-chat/`](packages/ai-chat) | AI chat layer | +| [`packages/hono-agents/`](packages/hono-agents) | Hono integration | +| [`packages/codemode/`](packages/codemode) | Code Mode (experimental) | +| [`examples/`](examples) | Self-contained demo apps | +| [`openai-sdk/`](openai-sdk) | Examples using the OpenAI Agents SDK | +| [`guides/`](guides) | In-depth pattern tutorials | +| [`docs/`](docs) | Markdown docs synced to developers.cloudflare.com | +| [`site/`](site) | Deployed websites (agents.cloudflare.com, AI playground) | +| [`design/`](design) | Architecture and design decision records | +| [`scripts/`](scripts) | Repo-wide tooling | + +## Development + +Node 24+ required. Uses npm workspaces. + +```sh +npm install # install all workspaces +npm run build # build all packages +npm run check # full CI check (format, lint, typecheck, exports) +CI=true npm test # run tests (vitest + vitest-pool-workers) +``` + +Changes to `packages/` need a changeset: + +```sh +npx changeset +``` + +See [`AGENTS.md`](AGENTS.md) for deeper contributor guidance. + +## Contributing + +We are not accepting external pull requests at this time — the SDK is evolving quickly and we want to keep the surface area manageable. That said, we'd love to hear from you: + +- **Bug reports & feature requests** — [open an issue](https://github.com/cloudflare/agents/issues) +- **Questions & ideas** — [start a discussion](https://github.com/cloudflare/agents/discussions) + +## License + +[MIT](LICENSE) diff --git a/THIRD_PARTY_LICENSES.md b/THIRD_PARTY_LICENSES.md new file mode 100644 index 0000000000..6470f3d0bb --- /dev/null +++ b/THIRD_PARTY_LICENSES.md @@ -0,0 +1,149 @@ +# Third-Party Licenses + +This project incorporates code from the following open-source projects. + +## Dependencies + +### Anthropic MCP TS SDK + +- **License**: Apache License 2.0 +- **Repository**: https://github.com/modelcontextprotocol/typescript-sdk +- **Copyright**: Copyright 2024 Anthropic, PBC +- **Full License**: [licenses/mit-anthropic-mcp-ts-sdk.txt](licenses/mit-anthropic-mcp-ts-sdk.txt) + +### Coinbase x402 + +- **License**: Apache License 2.0 +- **Repository**: https://github.com/coinbase/x402 +- **Copyright**: Copyright 2024 Coinbase, Inc. +- **Full License**: [licenses/apache-2.0-coinbase-x402.txt](licenses/apache-2.0-coinbase-x402.txt) +- **Notice**: + > Apache-2.0 License + > + > Copyright 2024 Coinbase + > + > Licensed under the Apache License, Version 2.0 (the "License"); + > you may not use this file except in compliance with the License. + > You may obtain a copy of the License at + > + > http://www.apache.org/licenses/LICENSE-2.0 + > + > Unless required by applicable law or agreed to in writing, software + > distributed under the License is distributed on an "AS IS" BASIS, + > WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + > See the License for the specific language governing permissions and + > limitations under the License. + +### ethanniser/x402-mcp + +- **License**: MIT License +- **Repository**: https://github.com/ethanniser/x402-mcp +- **Copyright**: Copyright (c) 2025 Ethan Niser +- **Full License**: [licenses/mit-ethanniser-x402-mcp.txt](licenses/mit-ethanniser-x402-mcp.txt) + +### Vercel AI SDK + +- **License**: Apache License 2.0 +- **Repository**: https://github.com/vercel/ai +- **Copyright**: Copyright 2023 Vercel, Inc. +- **Full License**: [licenses/apache-2.0-vercel-ai-sdk.txt](licenses/apache-2.0-vercel-ai-sdk.txt) +- **Notice**: + > Apache-2.0 License + > + > Copyright 2023 Vercel, Inc. + > + > Licensed under the Apache License, Version 2.0 (the "License"); + > you may not use this file except in compliance with the License. + > You may obtain a copy of the License at + > + > http://www.apache.org/licenses/LICENSE-2.0 + > + > Unless required by applicable law or agreed to in writing, software + > distributed under the License is distributed on an "AS IS" BASIS, + > WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + > See the License for the specific language governing permissions and + > limitations under the License. + +### TypeScript + +- **License**: Apache License 2.0 +- **Repository**: https://github.com/microsoft/typescript +- **Copyright**: Copyright (c) Microsoft Corporation +- **Full License**: [licenses/apache-2.0-typescript.txt](licenses/apache-2.0-typescript.txt) +- **Notice**: + > Apache License + > + > Version 2.0, January 2004 + > + > http://www.apache.org/licenses/ + > + > TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + > + > 1. Definitions. + > + > "License" shall mean the terms and conditions for use, reproduction, and distribution as defined by Sections 1 through 9 of this document. + > + > "Licensor" shall mean the copyright owner or entity authorized by the copyright owner that is granting the License. + > + > "Legal Entity" shall mean the union of the acting entity and all other entities that control, are controlled by, or are under common control with that entity. For the purposes of this definition, "control" means (i) the power, direct or indirect, to cause the direction or management of such entity, whether by contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial ownership of such entity. + > + > "You" (or "Your") shall mean an individual or Legal Entity exercising permissions granted by this License. + > + > "Source" form shall mean the preferred form for making modifications, including but not limited to software source code, documentation source, and configuration files. + > + > "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types. + > + > "Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work (an example is provided in the Appendix below). + > + > "Derivative Works" shall mean any work, whether in Source or Object form, that is based on (or derived from) the Work and for which the editorial revisions, annotations, elaborations, or other modifications represent, as a whole, an original work of authorship. For the purposes of this License, Derivative Works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work and Derivative Works thereof. + > + > "Contribution" shall mean any work of authorship, including the original version of the Work and any modifications or additions to that Work or Derivative Works thereof, that is intentionally submitted to Licensor for inclusion in the Work by the copyright owner or by an individual or Legal Entity authorized to submit on behalf of the copyright owner. For the purposes of this definition, "submitted" means any form of electronic, verbal, or written communication sent to the Licensor or its representatives, including but not limited to communication on electronic mailing lists, source code control systems, and issue tracking systems that are managed by, or on behalf of, the Licensor for the purpose of discussing and improving the Work, but excluding communication that is conspicuously marked or otherwise designated in writing by the copyright owner as "Not a Contribution." + > + > "Contributor" shall mean Licensor and any individual or Legal Entity on behalf of whom a Contribution has been received by Licensor and subsequently incorporated within the Work. + > + > 2. Grant of Copyright License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable copyright license to reproduce, prepare Derivative Works of, publicly display, publicly perform, sublicense, and distribute the Work and such Derivative Works in Source or Object form. + > 3. Grant of Patent License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this section) patent license to make, have made, use, offer to sell, sell, import, and otherwise transfer the Work, where such license applies only to those patent claims licensable by such Contributor that are necessarily infringed by their Contribution(s) alone or by combination of their Contribution(s) with the Work to which such Contribution(s) was submitted. If You institute patent litigation against any entity (including a cross-claim or counterclaim in a lawsuit) alleging that the Work or a Contribution incorporated within the Work constitutes direct or contributory patent infringement, then any patent licenses granted to You under this License for that Work shall terminate as of the date such litigation is filed. + > 4. Redistribution. You may reproduce and distribute copies of the Work or Derivative Works thereof in any medium, with or without modifications, and in Source or Object form, provided that You meet the following conditions: + > + > You must give any other recipients of the Work or Derivative Works a copy of this License; and + > + > You must cause any modified files to carry prominent notices stating that You changed the files; and + > + > You must retain, in the Source form of any Derivative Works that You distribute, all copyright, patent, trademark, and attribution notices from the Source form of the Work, excluding those notices that do not pertain to any part of the Derivative Works; and + > + > If the Work includes a "NOTICE" text file as part of its distribution, then any Derivative Works that You distribute must include a readable copy of the attribution notices contained within such NOTICE file, excluding those notices that do not pertain to any part of the Derivative Works, in at least one of the following places: within a NOTICE text file distributed as part of the Derivative Works; within the Source form or documentation, if provided along with the Derivative Works; or, within a display generated by the Derivative Works, if and wherever such third-party notices normally appear. The contents of the NOTICE file are for informational purposes only and do not modify the License. You may add Your own attribution notices within Derivative Works that You distribute, alongside or as an addendum to the NOTICE text from the Work, provided that such additional attribution notices cannot be construed as modifying the License. You may add Your own copyright statement to Your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of Your modifications, or for any such Derivative Works as a whole, provided Your use, reproduction, and distribution of the Work otherwise complies with the conditions stated in this License. + > + > 5. Submission of Contributions. Unless You explicitly state otherwise, any Contribution intentionally submitted for inclusion in the Work by You to the Licensor shall be under the terms and conditions of this License, without any additional terms or conditions. Notwithstanding the above, nothing herein shall supersede or modify the terms of any separate license agreement you may have executed with Licensor regarding such Contributions. + > 6. Trademarks. This License does not grant permission to use the trade names, trademarks, service marks, or product names of the Licensor, except as required for reasonable and customary use in describing the origin of the Work and reproducing the content of the NOTICE file. + > 7. Disclaimer of Warranty. Unless required by applicable law or agreed to in writing, Licensor provides the Work (and each Contributor provides its Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied, including, without limitation, any warranties or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for determining the appropriateness of using or redistributing the Work and assume any risks associated with Your exercise of permissions under this License. + > 8. Limitation of Liability. In no event and under no legal theory, whether in tort (including negligence), contract, or otherwise, unless required by applicable law (such as deliberate and grossly negligent acts) or agreed to in writing, shall any Contributor be liable to You for damages, including any direct, indirect, special, incidental, or consequential damages of any character arising as a result of this License or out of the use or inability to use the Work (including but not limited to damages for loss of goodwill, work stoppage, computer failure or malfunction, or any and all other commercial damages or losses), even if such Contributor has been advised of the possibility of such damages. + > 9. Accepting Warranty or Additional Liability. While redistributing the Work or Derivative Works thereof, You may choose to offer, and charge a fee for, acceptance of support, warranty, indemnity, or other liability obligations and/or rights consistent with this License. However, in accepting such obligations, You may act only on Your own behalf and on Your sole responsibility, not on behalf of any other Contributor, and only if You agree to indemnify, defend, and hold each Contributor harmless for any liability incurred by, or claims asserted against, such Contributor by reason of your accepting any such warranty or additional liability. + > + > END OF TERMS AND CONDITIONS + +### TypeScript VFS + +- **License**: MIT License +- **Repository**: https://github.com/microsoft/TypeScript-Website/tree/v2/packages/typescript-vfs +- **Copyright**: Copyright (c) Microsoft Corporation +- **Full License**: [licenses/mit-typescript-vfs.txt](licenses/mit-typescript-vfs.txt) +- **Notice**: + > The MIT License (MIT) + > Copyright (c) Microsoft Corporation + > + > Permission is hereby granted, free of charge, to any person obtaining a copy of this software and + > associated documentation files (the "Software"), to deal in the Software without restriction, + > including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, + > and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, + > subject to the following conditions: + > + > The above copyright notice and this permission notice shall be included in all copies or substantial + > portions of the Software. + > + > THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT + > NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. + > IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, + > WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE + > SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +--- diff --git a/assets/npm-install-agents.svg b/assets/npm-install-agents.svg new file mode 100644 index 0000000000..d9bdfc0416 --- /dev/null +++ b/assets/npm-install-agents.svg @@ -0,0 +1,24 @@ + + + + + + + + + + + + + + + + + + $ + npm i agents + + + + + diff --git a/design/AGENTS.md b/design/AGENTS.md new file mode 100644 index 0000000000..d947540bf9 --- /dev/null +++ b/design/AGENTS.md @@ -0,0 +1,91 @@ +# AGENTS.md — design/ + +Internal design records — the "why" behind decisions in this repo and its libraries. This is the Diátaxis **explanation** quadrant: architecture rationale, tradeoffs, and alternatives considered. + +## Two kinds of document + +### Design docs + +Living documents that describe how a concept or subsystem works **right now**. Named by topic: `state.md`, `mcp.md`, `visuals.md`. These are the primary entry point — a contributor looking for "how does state work" should open one file and get the full picture. + +Design docs get updated as the implementation evolves. They always reflect the current reality. + +### RFCs + +Point-in-time decision records for significant changes. Named with an `rfc-` prefix: `rfc-state-v2-sync-protocol.md`. These capture why a specific change was made and what alternatives were considered. They do not get updated after the decision — they are snapshots. + +RFCs are never deleted, even after rejection. Rejected RFCs are valuable — they prevent re-litigating the same idea later. + +## Workflow + +``` +1. Propose: write rfc-.md (status: proposed) +2. Decide: update status to accepted or rejected +3. Implement: update the relevant design doc to reflect the new reality + (create one if it does not exist yet) +``` + +Step 3 is important — the design doc is what people read day-to-day. The RFC is the footnote explaining one particular decision within it. + +A design doc may link to multiple RFCs that shaped it over time: + +```markdown +## History + +- [rfc-state-sync.md](./rfc-state-sync.md) — original bidirectional sync design +- [rfc-state-v2-batching.md](./rfc-state-v2-batching.md) — added batched updates +``` + +## RFC format + +Include a status line at the top: + +``` +Status: proposed | accepted | rejected +``` + +Then cover: + +- **The problem** — what we need to solve +- **The proposal** — what we want to do +- **The alternatives** — what else we considered and why not +- **The decision** — what was decided (filled in after discussion) + +## Design doc format + +No strict template. Each file should at minimum cover: + +- **How it works** — the current design, kept up to date +- **Key decisions** — link to relevant RFCs for the reasoning +- **Tradeoffs** — what we gave up and why + +Keep it concise. A few paragraphs is fine. These are records, not essays. + +## What does not belong here + +- **API reference or usage guides** — those go in `/docs` (see `/docs/AGENTS.md`) +- **Code comments** — keep inline explanations in the code itself +- **Changelogs** — those live in package `CHANGELOG.md` files + +## Current contents + +| File | Type | Scope | +| ------------------------- | ---------- | ------------------------------------------------------------------------------------- | +| `chat-shared-layer.md` | design doc | Chat shared layer — streaming, sanitization, and protocol primitives in agents/chat | +| `think.md` | design doc | Think — chat agent base class, streaming, client tools, resumable streams, extensions | +| `think-sessions.md` | design doc | Think + Session integration design (implemented in Phase 1) | +| `think-vs-aichat.md` | design doc | Think vs AIChatAgent — comparison, use cases, architectural differences | +| `think-roadmap.md` | design doc | Think implementation plan — all 5 phases complete, full AIChatAgent parity | +| `chat-api.md` | analysis | AIChatAgent + useAgentChat API analysis — pain points, improvements, Think influence | +| `chat-improvements.md` | design doc | Non-breaking improvements — shared extraction complete, client DX items remain | +| `readonly-connections.md` | design doc | Readonly connections — enforcement, storage wrapping, caveats | +| `retries.md` | design doc | Retry system — primitives, integration points, backoff strategy, tradeoffs | +| `visuals.md` | design doc | UI component library (Kumo), dark mode, custom patterns, routing integration | +| `workspace.md` | design doc | Workspace — hybrid SQLite+R2 filesystem, bash, symlinks, observability | +| `rfc-sub-agents.md` | RFC | Sub-agents — child DOs via facets, typed stubs, built into Agent (accepted) | +| `loopback.md` | design doc | Loopback pattern — cross-boundary RPC for sub-agents and dynamic isolates | +| `worker-bundler.md` | design doc | Worker bundler — host-side assets, no code generation, mounting is caller's concern | + +## Relationship to `/docs` + +`/docs` is user-facing ("how to use the SDK"). `/design` is contributor-facing ("why the SDK works this way"). If a design decision affects how users interact with the SDK, distil the user-relevant parts into a doc in `/docs` and link back here for the full rationale. diff --git a/design/README.md b/design/README.md new file mode 100644 index 0000000000..1345a04d9b --- /dev/null +++ b/design/README.md @@ -0,0 +1,16 @@ +# Design + +This folder documents design decisions, tradeoffs, and rationale across the Agents SDK repository and its libraries. It covers software architecture, API design, visual/UI choices, and anything else where we made a deliberate decision worth recording. + +The goal is to give contributors (and future-us) a quick way to understand _why_ things are the way they are, without having to reverse-engineer intent from code or PR history. + +## Contents + +| File | Scope | +| ---------------------------------------------------- | --------------------------------------------------------------------------- | +| [think.md](./think.md) | Think — chat agent base class, sessions, streaming, tools, execution ladder | +| [visuals.md](./visuals.md) | UI component library choice, Kumo usage, custom patterns | +| [readonly-connections.md](./readonly-connections.md) | Readonly connection enforcement, storage, tradeoffs, and caveats | +| [workspace.md](./workspace.md) | Workspace — hybrid SQLite+R2 filesystem, bash, symlinks | +| [rfc-sub-agents.md](./rfc-sub-agents.md) | RFC: Sub-agents — child DOs via facets, typed stubs, mixin API | +| [loopback.md](./loopback.md) | Loopback pattern — cross-boundary RPC for sub-agents and dynamic isolates | diff --git a/design/browser-tools.md b/design/browser-tools.md new file mode 100644 index 0000000000..e2314b45dd --- /dev/null +++ b/design/browser-tools.md @@ -0,0 +1,48 @@ +# Browser Tools + +**Status:** experimental (`agents/browser`) + +## Problem + +The browser tools need to expose the Chrome DevTools Protocol without shipping a large generated protocol bundle in the package, and they need to work the same way in local development and deployed environments. + +## How It Works + +The browser tools expose Chrome DevTools Protocol (CDP) access through the same code-mode pattern used elsewhere in the SDK. The LLM never talks to the browser directly. + +`browser_search` fetches the protocol description from a live CDP endpoint, normalizes it into the compact search shape the tool prompt expects, and exposes that normalized spec to sandboxed code through `spec.get()`. + +`browser_execute` opens a browser-level CDP WebSocket from the host Worker, then exposes a narrow RPC surface (`cdp.send()`, `cdp.attachToTarget()`, debug-log helpers) to sandboxed code. The sandbox issues RPC calls, while the actual WebSocket and browser session remain host-side. + +When the Browser Rendering binding is available, the host uses the binding's devtools endpoints directly: + +- `POST /v1/devtools/browser` to acquire a short-lived session for protocol fetches +- `GET /v1/devtools/browser` with `Upgrade: websocket` to acquire and connect for execution +- `GET /v1/devtools/browser/:sessionId/json/protocol` to read the live protocol +- `DELETE /v1/devtools/browser/:sessionId` to release the session + +`cdpUrl` remains as an override for custom CDP endpoints, but it is no longer the primary local-development path. + +## Key Decisions + +- Do not bundle the CDP spec in the package. The browser already exposes it, and fetching it at runtime avoids shipping a large generated asset plus a spec-generation build step. +- Normalize the live protocol before exposing it to the sandbox. The raw `/json/protocol` payload uses Chrome's schema (`domain`, command `name`, optional arrays). The tool prompt and user code are simpler when they always receive `name`, `method` or `event`, and concrete arrays. +- Keep browser state on the host side. The sandbox gets only RPC helpers, which preserves the existing code-mode isolation model and avoids giving arbitrary browser or network access to LLM-generated code. +- Use a fresh browser session per `browser_execute` call. This keeps the lifecycle simple and makes cleanup deterministic, at the cost of cross-call session continuity. + +## Tradeoffs + +- Fetching the spec at runtime adds a small amount of latency compared with a bundled JSON file, so the implementation keeps a short in-memory cache. +- The cache is intentionally shallow and process-local. It reduces repeated fetches during a session without introducing persistence, invalidation complexity, or versioned assets. +- Per-call browser sessions make reasoning about cleanup easy, but longer workflows need to fit inside one tool invocation. +- The Browser Rendering binding is now the default local-development path, which is much simpler for users, but it means local behavior depends on the Wrangler/Miniflare version providing the devtools endpoints. + +## Verification + +The end-to-end tests start a real `wrangler dev` instance, talk to an agent over WebSocket RPC, and exercise both tools against the local Browser Rendering binding. The suite covers: + +- protocol search against the live `/json/protocol` endpoint +- simple browser commands such as `Browser.getVersion` +- page creation and navigation +- a DOM or runtime read after navigation +- representative error handling paths diff --git a/design/chat-api.md b/design/chat-api.md new file mode 100644 index 0000000000..e44d83e677 --- /dev/null +++ b/design/chat-api.md @@ -0,0 +1,781 @@ +# Chat API Analysis: AIChatAgent + useAgentChat + +A critical analysis of the current `@cloudflare/ai-chat` API surface — both the server-side `AIChatAgent` class and the client-side `useAgentChat` React hook. Identifies pain points, awkward patterns, missing capabilities, and opportunities for improvement. + +> **Note:** The "Implications for Think" section was written before the Session integration design. For how Think addresses these issues with Session as its storage layer, see [think-roadmap.md](./think-roadmap.md). Several server-side issues (S3 message access helpers, X2 conversation management) are resolved by Session. The client-side issues (C1–C8) remain fully relevant — they affect `useAgentChat` regardless of server-side architecture. + +Related: + +- [think-vs-aichat.md](./think-vs-aichat.md) — feature gap analysis between Think and AIChatAgent +- [think-sessions.md](./think-sessions.md) — Session integration design for Think +- [think-roadmap.md](./think-roadmap.md) — implementation plan (supersedes the prioritization in this doc) + +--- + +## Table of Contents + +1. [Current API Surface](#current-api-surface) +2. [What Works Well](#what-works-well) +3. [Server-Side Issues](#server-side-issues) +4. [Client-Side Issues](#client-side-issues) +5. [Cross-Cutting Issues](#cross-cutting-issues) +6. [Implications for Think](#implications-for-think) +7. [Summary: Prioritized Improvements](#summary-prioritized-improvements) + +--- + +## Current API Surface + +### Server: `AIChatAgent` (`@cloudflare/ai-chat`) + +Core class that extends `Agent` from the Agents SDK. Users subclass it and override `onChatMessage`: + +```typescript +export class MyAgent extends AIChatAgent { + async onChatMessage( + onFinish: StreamTextOnFinishCallback, + options?: OnChatMessageOptions + ): Promise { + const result = streamText({ + model: createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ), + system: "You are a helpful assistant.", + messages: pruneMessages({ + messages: await convertToModelMessages(this.messages), + toolCalls: "before-last-2-messages" + }), + tools: { + /* ... */ + }, + abortSignal: options?.abortSignal + }); + return result.toUIMessageStreamResponse(); + } +} +``` + +**Override points and properties:** + +| Member | Type | Purpose | +| ------------------------------------ | ------------------------ | ----------------------------------- | +| `onChatMessage(onFinish, options?)` | override | Handle chat turn, return `Response` | +| `onChatResponse(result)` | override | Post-turn lifecycle hook | +| `onChatRecovery(ctx)` | override | Recovery policy after DO eviction | +| `sanitizeMessageForPersistence(msg)` | override | Custom pre-persist transform | +| `this.messages` | `ChatMessage[]` | Current conversation history | +| `maxPersistedMessages` | `number \| undefined` | Storage cap | +| `messageConcurrency` | `MessageConcurrency` | Overlap strategy | +| `chatRecovery` | `boolean` | Fiber-wrapped turns | +| `waitForMcpConnections` | `boolean \| { timeout }` | MCP wait | +| `saveMessages(msgs)` | method | Programmatic turn | +| `continueLastTurn(body?)` | method | Continue last assistant message | +| `persistMessages(msgs)` | method | Persist + broadcast | +| `hasPendingInteraction()` | method | Client tool pending check | +| `waitUntilStable(opts?)` | method | Await quiescence | +| `resetTurnState()` | method | Abort + reset | + +**`OnChatMessageOptions`:** + +```typescript +export type OnChatMessageOptions = { + requestId: string; + abortSignal?: AbortSignal; + clientTools?: ClientToolSchema[]; + body?: Record; +}; +``` + +**`ChatResponseResult` (for `onChatResponse`):** + +```typescript +export type ChatResponseResult = { + message: ChatMessage; + requestId: string; + continuation: boolean; + status: "completed" | "error" | "aborted"; + error?: string; +}; +``` + +### Client: `useAgentChat` (`@cloudflare/ai-chat/react`) + +React hook that wraps the AI SDK's `useChat` with WebSocket transport and agent-specific features: + +```typescript +const agent = useAgent({ agent: "ChatAgent", name: "session-1" }); +const { + messages, + sendMessage, + clearHistory, + stop, + addToolOutput, + addToolApprovalResponse, + status, + isServerStreaming, + isStreaming +} = useAgentChat({ agent }); +``` + +**`UseAgentChatOptions` (agent-specific fields, beyond `useChat` options):** + +| Option | Type | Default | Purpose | +| --------------------------------------- | ---------------------------------- | ------------- | ----------------------------------------- | +| `agent` | `ReturnType` | required | WebSocket connection | +| `getInitialMessages` | `((opts) => Promise) \| null` | default fetch | Custom initial message source | +| `credentials` | `RequestCredentials` | — | HTTP fetch credentials | +| `headers` | `HeadersInit` | — | HTTP fetch headers | +| `onToolCall` | `OnToolCallCallback` | — | Client-side tool execution | +| `tools` | `Record` | — | ~~Dynamic client tools~~ (deprecated) | +| `toolsRequiringConfirmation` | `string[]` | — | ~~Manual confirmation list~~ (deprecated) | +| `experimental_automaticToolResolution` | `boolean` | — | ~~Auto-resolve tools~~ (deprecated) | +| `autoContinueAfterToolResult` | `boolean` | `true` | Server auto-continues after tool result | +| `autoSendAfterAllConfirmationsResolved` | `boolean` | `true` | ~~Client batching~~ (deprecated) | +| `resume` | `boolean` | `true` | Stream resumption on reconnect | +| `body` | `object \| (() => object)` | — | Custom data in every request | +| `prepareSendMessagesRequest` | `(opts) => { body?, headers? }` | — | Advanced request customization | + +**Return type additions (beyond `useChat`):** + +| Field | Type | Purpose | +| ------------------- | ---------------- | ------------------------------------ | +| `clearHistory` | `() => void` | Clear conversation | +| `addToolOutput` | `(opts) => void` | Provide tool result | +| `isServerStreaming` | `boolean` | Server-initiated stream active | +| `isStreaming` | `boolean` | Any stream active (client or server) | + +### Transport: `WebSocketChatTransport` + +Implements the AI SDK's `ChatTransport` interface over WebSocket. Not typically used directly by app code: + +```typescript +interface ChatTransport { + sendMessages(options: { messages: M[]; ... }): Promise>; + reconnectToStream(options: { chatId: string }): Promise | null>; +} +``` + +The transport maps between `CF_AGENT_USE_CHAT_REQUEST` / `CF_AGENT_USE_CHAT_RESPONSE` WebSocket frames and the `ReadableStream` that `useChat` consumes. It handles: + +- Request ID correlation +- Abort/cancel signaling +- Stream resume handshake +- Tool continuation streams + +### Wire Protocol + +See [chat-shared-layer.md](./chat-shared-layer.md) and the `types.ts` `MessageType` enum. The full protocol includes 12 message types for chat requests/responses, message sync, stream resume, tool results/approvals, and clear commands. + +--- + +## What Works Well + +### Wire protocol design + +The `cf_agent_chat_*` WebSocket protocol is well-designed and battle-tested: + +- **Resumable streams**: Chunk buffering in SQLite + replay on reconnect is a genuine differentiator. The 3-step handshake (`STREAM_RESUMING` → `STREAM_RESUME_ACK` → replay chunks) correctly handles reconnection timing, and `_pendingResumeConnections` prevents duplicate chunks during replay. +- **Multi-tab broadcast**: Chunks are broadcast to all connections. The `activeRequestIds` mechanism in the transport correctly deduplicates, while `onAgentMessage` in the hook handles cross-tab updates via `broadcastTransition`. +- **Continuation protocol**: The `continuation: true` flag on response frames + `messageId` stripping enables clean message append semantics. + +### Tool execution model + +The preferred pattern (`onToolCall` callback) is clean and well-structured: + +```typescript +useAgentChat({ + agent, + onToolCall: async ({ toolCall, addToolOutput }) => { + if (toolCall.toolName === "getLocation") { + const pos = await navigator.geolocation.getCurrentPosition(); + addToolOutput({ + toolCallId: toolCall.toolCallId, + output: { lat: pos.coords.latitude, lng: pos.coords.longitude } + }); + } + } +}); +``` + +The server-side `needsApproval` + client-side `addToolApprovalResponse` flow for human-in-the-loop approval is also well-structured, with the server persisting the approval state and broadcasting `CF_AGENT_MESSAGE_UPDATED` when it changes. + +### `body` option + +The `body` option on `useAgentChat` — supporting both static objects and dynamic functions — is well-designed: + +```typescript +// Static +body: { timezone: "America/New_York", userId: "abc" } + +// Dynamic (called on each send) +body: () => ({ token: getAuthToken(), timestamp: Date.now() }) +``` + +On the server, `options.body` is persisted to SQLite and survives hibernation, so tool continuations and recovery flows inherit the original request context. The `prepareSendMessagesRequest` escape hatch provides additional flexibility for advanced cases. + +### `autoContinueAfterToolResult` + +The default `true` behavior for auto-continuation after tool results mirrors how server-executed tools work with `maxSteps` in `streamText`. The 10ms coalesce window batches rapid tool results into a single continuation turn. This is the right default — most apps want the LLM to respond after tool results without manual intervention. + +--- + +## Server-Side Issues + +### Issue S1: `onChatMessage` signature is awkward + +**Current:** + +```typescript +async onChatMessage( + onFinish: StreamTextOnFinishCallback, + options?: OnChatMessageOptions +): Promise +``` + +**Problems:** + +1. **`onFinish` is almost never used.** The framework passes `async (_finishResult) => {}` (a no-op) at every call site — WebSocket turns, auto-continuations, programmatic turns, and `continueLastTurn`. Every example in the repo either ignores it or passes it through without using the data. It's a vestige of an earlier design where `onFinish` was the persistence hook; now `_reply` handles persistence internally. + +2. **The `Response` return type is an abstraction mismatch.** AIChatAgent communicates over WebSocket, not HTTP. The `Response` is never sent to any HTTP client — it's consumed internally by `_reply`, which reads the body stream and detects content type (`text/event-stream` vs plaintext). Users must call `.toUIMessageStreamResponse()` even though no HTTP response is sent. Content-type detection and HTTP status codes are HTTP concepts applied to an internal pipeline. + +3. **`undefined` return has no ergonomic use.** When `onChatMessage` returns `undefined`, the framework logs a warning and sends a terminal `done: true` frame with "No response was generated by the agent." This is almost always a bug, not intentional. A void return should either be an error or have clear semantics (e.g., "I handled this myself via `broadcast`"). + +**Compare Think:** + +```typescript +async onChatMessage(options?: ChatMessageOptions): Promise +``` + +Think's signature is cleaner: no `onFinish` parameter, and the return type (`StreamableResult` with `toUIMessageStream()`) matches what's actually needed — an async iterable of chunks, not an HTTP response. + +**Recommendation:** Drop `onFinish` from the signature. If users need provider finish metadata, they can get it from `onChatResponse` (which already provides richer context) or wire their own `onFinish` into `streamText` inside their override. Accept `StreamableResult | Response` for backward compat. + +### Issue S2: `onChatMessage` doesn't know if it's a continuation + +`OnChatMessageOptions` has no `continuation` field. When the server auto-continues after tool results, or when `continueLastTurn()` fires after recovery, `onChatMessage` is called identically to a fresh user turn. + +The only signals available to subclasses are: + +- `options.requestId` — client-generated for user submits, server-generated `nanoid()` for continuations (but this is an implementation detail, not a semantic signal) +- Inspecting `this.messages` for patterns (last message is assistant with pending tool state) +- `onChatResponse`'s `result.continuation` — but that fires _after_ the turn, not during + +The `experimental/forever-chat` example works around this with `options?.body?.recovering` — a client-side flag that's fragile and requires client-server coordination for what should be a server-internal concept. + +**Recommendation:** Add `continuation: boolean` to `OnChatMessageOptions`. This lets subclasses: + +- Adjust system prompts for continuations vs fresh turns +- Select different models (faster model for continuations) +- Skip expensive context assembly (RAG, memory retrieval) on continuations +- Log continuation vs initial turn metrics + +### Issue S3: `this.messages` is the only context access + +The server-side API relies entirely on `this.messages` (a flat `ChatMessage[]`) for conversation history. There are no helper methods for common access patterns: + +```typescript +// Every app does this: +const lastAssistant = this.messages.filter((m) => m.role === "assistant").pop(); +const recentMessages = this.messages.slice(-10); +const hasPendingTool = this.messages.some( + (m) => + m.role === "assistant" && + m.parts.some((p) => "state" in p && p.state === "input-available") +); +``` + +Think partially addresses this with `assembleContext()` (a structured override for context customization) and `getMessages()`, but the underlying data access is still raw array manipulation. + +**Recommendation:** Add helper methods: + +- `getLastAssistantMessage(): ChatMessage | undefined` +- `getLastUserMessage(): ChatMessage | undefined` +- `getRecentMessages(n: number): ChatMessage[]` +- `getMessageById(id: string): ChatMessage | undefined` +- Consider a `ConversationContext` object passed to `onChatMessage` that provides these + +### Issue S4: `Response` return type couples to HTTP semantics + +As noted in S1, the `Response` return type is a leaky abstraction. Beyond the ergonomic issue, it has practical consequences: + +1. **Content-type detection**: `_reply` checks `response.headers.get("content-type")` to choose between SSE parsing and plaintext wrapping. This is an HTTP concept being used as a dispatch mechanism inside a WebSocket pipeline. + +2. **Two parsing paths**: The SSE path (`_streamSSEReply`) and the plaintext path (`_sendPlaintextReply`) have different message building, error handling, and continuation semantics. This dual-path complexity exists because `Response` is too generic — it could be anything. + +3. **No type safety**: A `Response` provides no compile-time guarantee about what's inside. Users could return `new Response("oops")` or `new Response(JSON.stringify({foo: "bar"}))` and the framework would attempt to parse it. + +Think's `StreamableResult` (an object with `toUIMessageStream(): AsyncIterable`) is closer to what's actually needed, though it loses the plaintext convenience. + +**Recommendation:** Define a proper return type: + +```typescript +type ChatTurnResult = + | { stream: AsyncIterable } + | { text: string } + | Response; // backward compat +``` + +Or simply accept `StreamableResult | Response` and detect via duck typing. + +### Issue S5: Persistence timing is opaque + +The relationship between `onChatMessage` and persistence is complex and not well-documented: + +1. **User messages** are persisted _before_ `onChatMessage` is called (inside `persistMessages` during the WebSocket handler). +2. **Assistant messages** are persisted _after_ the stream completes (inside `_reply`). +3. **Tool approval states** get an early SQL write _during_ streaming (when a tool hits `approval-requested` state) to survive refresh. +4. **`saveMessages`** persists first, then runs `onChatMessage`. + +Subclasses have no control over when or how assistant messages are persisted. If a subclass wants to modify the assistant message before persistence (e.g., add metadata), it must use `sanitizeMessageForPersistence` — but that runs after the stream is complete and the message is fully assembled. There's no hook for _during_ streaming. + +**Recommendation:** Document the persistence timeline clearly. Consider exposing a `beforePersist(message)` hook or integrating `sanitizeMessageForPersistence` into Think's pipeline. + +--- + +## Client-Side Issues + +### Issue C1: Suspense-only initial message fetch + +`useAgentChat` uses React's `use()` to suspend while fetching initial messages from `/get-messages`: + +```typescript +// packages/ai-chat/src/react.tsx +const initialMessages = initialMessagesPromise + ? use(initialMessagesPromise) // ← suspends here + : (optionsInitialMessages ?? []); +``` + +This creates several problems documented in [#1011](https://github.com/cloudflare/agents/issues/1011) and [#1045](https://github.com/cloudflare/agents/issues/1045): + +**Problem 1: No non-suspending option.** There's no `isPending` / `error` return state. Components must be wrapped in ``, and the Suspense boundary location must match the fetch location. This prevents patterns like: + +- Fetching in a parent, showing loading in a child +- Keeping the fetch alive when a panel toggles (unmount kills the `use()` suspension) +- Fine-grained loading states (skeleton messages vs full-page spinner) + +**Problem 2: No exported fetch function.** The `defaultGetInitialMessagesFetch` function is internal: + +```typescript +// packages/ai-chat/src/react.tsx, lines 495–520 +async function defaultGetInitialMessagesFetch({ url }: GetInitialMessagesOptions) { + const getMessagesUrl = new URL(url); + getMessagesUrl.pathname += "/get-messages"; + const response = await fetch(getMessagesUrl.toString(), { ... }); + return JSON.parse(text) as ChatMessage[]; +} +``` + +Framework route loaders (Next.js, TanStack Router, Remix) can't prefetch messages during route loading. Users must reconstruct the URL manually (`{host}/agents/{agent}/{name}/get-messages`) and pass `getInitialMessages: null` + `messages`, which depends on knowing the internal URL format and isn't a supported pattern. + +**Problem 3: No `fallbackMessages`.** Switching between conversations always suspends, even if the user just visited that conversation. There's no SWR-style `fallbackData` / `placeholderData` pattern to show cached messages while revalidating. The issue at [#1045](https://github.com/cloudflare/agents/issues/1045) proposes this pattern: + +```typescript +useAgentChat({ + agent, + fallbackMessages: messageCache.get(conversationId) +}); +``` + +The request cache (`requestCache` Map at module level) provides deduplication across React Strict Mode double-renders, but not stale-while-revalidate semantics. + +**Problem 4: No error boundary integration.** If the fetch fails, `use()` throws and must be caught by an Error Boundary. There's no graceful degradation or retry mechanism. + +**Recommendation:** + +- Export `getAgentMessages({ host, agent, name, credentials?, headers? })` as a standalone fetch function +- Add a non-suspending variant that returns `{ messages, isPending, error }` (like TanStack Query's `useQuery`) +- Support `fallbackMessages` for instant conversation switching with background revalidation +- Optionally rename the current suspending version to `useSuspenseAgentChat` (TanStack Query convention) + +### Issue C2: Two hook calls required for basic setup + +Every app needs both `useAgent` and `useAgentChat`: + +```typescript +const agent = useAgent({ + agent: "ChatAgent", + name: `session-${userId}`, + onOpen: () => setStatus("connected"), + onClose: () => setStatus("disconnected"), + onError: () => setStatus("disconnected") +}); + +const { messages, sendMessage, clearHistory, status } = useAgentChat({ + agent, + onToolCall: async ({ toolCall, addToolOutput }) => { + /* ... */ + } +}); +``` + +This two-step pattern makes sense architecturally (separate connection from chat), but adds friction for the 90% case. The `useAgent` return value is a `PartySocket` with extra fields (`state`, `setState`, `call`, `stub`) — implementation details that leak into every chat app. + +The hook also uses `@ts-expect-error` to access internal `PartySocket` properties: + +```typescript +// packages/ai-chat/src/react.tsx, lines 469–474 +const agentUrl = new URL( + `${// @ts-expect-error we're using a protected _url property + ((agent._url as string | null) || agent._pkurl) + ?.replace("ws://", "http://") + .replace("wss://", "https://")}` +); +``` + +This coupling to `PartySocket` internals is fragile and creates maintenance burden. + +**Recommendation:** Provide a combined `useAgentChat({ agent: "ChatAgent", name: "session-1", host })` that manages the connection internally. Keep the split hooks available for advanced use cases (RPC, state management, non-chat agents), but offer a shorthand for the common case. Think could provide `useThinkChat` as its version of this. + +### Issue C3: Tool UI is entirely user-rebuilt every time + +Every example in the repo reimplements the same tool-state rendering logic. Tool parts have ~8 states, and every app handles them manually: + +```typescript +// This pattern appears in every example +for (const part of message.parts) { + if (part.type === "text") { + return {part.text}; + } + if (isToolUIPart(part)) { + if (part.state === "output-available") { + return
{JSON.stringify(part.output, null, 2)}
; + } + if (part.state === "input-available") { + return
Running {getToolName(part)}...
; + } + if (part.state === "approval-requested") { + return
+ + +
; + } + if (part.state === "input-streaming") { /* ... */ } + if (part.state === "output-error") { /* ... */ } + if (part.state === "output-denied") { /* ... */ } + // ... etc + } + if (part.type === "reasoning") { /* ... */ } +} +``` + +The playground's embedded documentation snippets are also stale — they show `toDataStreamResponse`, positional `useAgentChat(agent, ...)`, and `addToolResult` while the real code uses `toUIMessageStreamResponse`, `{ agent, ... }`, and `addToolOutput`. This creates confusion for developers reading the examples. + +**Recommendation:** + +- Export a `` component (or `useMessageParts` hook) that handles the state machine with render prop / slot-based customization +- At minimum, export a `getToolPartState(part): "loading" | "streaming" | "complete" | "error" | "denied" | "waiting-approval" | "approved"` utility that simplifies the state detection +- Consider a `` component with sensible defaults and customization slots (header, body, actions, error) + +### Issue C4: `isServerStreaming` vs `status` is confusing + +`useAgentChat` returns three streaming indicators: + +| Field | Source | Tracks | +| ------------------- | ---------------- | ---------------------------------------------------------------------- | +| `status` | AI SDK `useChat` | Client-initiated requests only | +| `isServerStreaming` | Agent hook | Server-initiated streams (saveMessages, auto-continuation, other tabs) | +| `isStreaming` | Derived | `status === "streaming" \|\| isServerStreaming` | + +Users consistently just want to know "is something streaming?" but must understand the distinction. The playground examples use `status === "streaming"` and miss server-initiated streams entirely. Other examples use `isStreaming` correctly but inconsistently. + +The AI SDK's `status` field has four values: `"ready"`, `"submitted"`, `"streaming"`, `"error"`. None of these reflect server-initiated activity. When `saveMessages` triggers a turn, the client sees chunks via `broadcastTransition` and `isServerStreaming` goes `true`, but `status` stays `"ready"`. + +**Recommendation:** Make `isStreaming` the primary documented API. Consider deprecating direct `status` inspection for streaming detection, or provide clear documentation about when each indicator is meaningful. Alternatively, unify into a single `chatStatus` that covers all states: + +```typescript +type ChatStatus = + | "idle" // No activity + | "submitting" // Client request in flight + | "streaming" // Any source streaming (client or server) + | "awaiting-tool" // Waiting for client tool result or approval + | "error"; // Last turn errored +``` + +### Issue C5: No structured error handling + +When `onChatMessage` throws on the server, the client receives an error frame: + +```typescript +// Server sends: +{ type: "cf_agent_use_chat_response", id, body: error.message, done: true, error: true } +``` + +On the client, this either: + +- Errors the transport's `ReadableStream` (for client-initiated requests) +- Gets processed by `broadcastTransition` with `error: true` (for cross-tab/server streams) + +But there's no structured `onChatError` callback in `useAgentChat`. The AI SDK's `useChat` has an `error` state in its return value, but it's unclear how agent-specific errors map to that vs. network errors vs. abort errors. + +**Recommendation:** Add `onChatError?: (error: { message: string; requestId: string; source: "server" | "network" }) => void` to `UseAgentChatOptions`. Ensure `error` in the return value distinguishes between chat errors and transport errors. + +### Issue C6: `addToolOutput` vs `addToolResult` naming confusion + +The hook exposes `addToolOutput` (which calls `addToolResult` internally). The AI SDK exposes `addToolResult` and `addToolApprovalResponse`. The naming is inconsistent: + +- `useAgentChat` return: `addToolOutput`, `addToolApprovalResponse` +- AI SDK `useChat` return: `addToolResult`, `addToolApprovalResponse` +- Wire protocol: `CF_AGENT_TOOL_RESULT` +- Server-side: `_applyToolResult` + +The `addToolOutput` wrapper exists because it also sends the result to the server (AI SDK's `addToolResult` only updates local state). But the naming difference creates confusion — users reading AI SDK docs find `addToolResult`, then discover `useAgentChat` has `addToolOutput` instead. + +**Recommendation:** Either align naming with the AI SDK (`addToolResult` that auto-sends to server), or clearly document the distinction. The type `AddToolOutputOptions` also has fields (`state`, `errorText`) that `addToolResult` doesn't, which further diverges the mental models. + +### Issue C7: Module-level `requestCache` leaks across instances + +The initial message cache is a module-level `Map`: + +```typescript +// packages/ai-chat/src/react.tsx, line 326 +const requestCache = new Map>(); +``` + +This is intentional (deduplication across React Strict Mode), but it has side effects: + +- Cache keys include `origin + pathname + agent + name` but not `credentials` or `headers`, so different auth contexts share cache entries +- No TTL — cached promises live until the component unmounts and the cleanup effect runs +- In SSR/RSC environments, module-level state persists across requests (request-scoping issues) + +**Recommendation:** Consider a WeakRef-based cache or a configurable cache provider. For SSR, the cache should be request-scoped. + +### Issue C8: `@ts-expect-error` coupling to PartySocket internals + +As noted in C2, `useAgentChat` accesses internal `PartySocket` properties (`_url`, `_pkurl`) to derive the HTTP URL for `/get-messages`: + +```typescript +// @ts-expect-error we're using a protected _url property +(agent._url as string | null) || agent._pkurl; +``` + +This creates fragile coupling to the `PartySocket` implementation. If the internal API changes, `useAgentChat` breaks silently at runtime. + +**Recommendation:** Add a public method to `useAgent`'s return value: `getHttpUrl(): string` (or `httpUrl` property). This encapsulates the URL derivation and removes the `@ts-expect-error`. + +--- + +## Cross-Cutting Issues + +### Issue X1: Deprecated options accumulating + +The hook has accumulated several deprecated options that add cognitive load and increase bundle size: + +| Deprecated Option | Replacement | Status | +| --------------------------------------- | ----------------------------------- | ----------------- | +| `tools` (with `execute`) | `onToolCall` callback | Still functional | +| `experimental_automaticToolResolution` | `onToolCall` | Still functional | +| `toolsRequiringConfirmation` | `needsApproval` on server tools | Still functional | +| `autoSendAfterAllConfirmationsResolved` | `sendAutomaticallyWhen` from AI SDK | Still functional | +| `JSONSchemaType` | `JSONSchema7` from `"ai"` | Re-exported alias | +| `detectToolsRequiringConfirmation()` | `needsApproval` | Exported function | +| `AITool.inputSchema` | `AITool.parameters` | Warning on use | + +The deprecated `experimental_automaticToolResolution` codepath alone is ~120 lines of `useEffect` logic (lines 864–992) that processes tool calls, sends results to server, manages local state, and handles re-entrancy — all duplicating what `onToolCall` does more cleanly. + +**Recommendation:** Plan a major version that removes these. In the interim, consider moving deprecated functionality to `@cloudflare/ai-chat/compat` to reduce the main bundle. The `tools` option (client-defined tool schemas) still has valid use cases in SDKs/platforms with dynamic tools — consider keeping it but renaming or restructuring. + +### Issue X2: No conversation / session management + +Both AIChatAgent and Think support only one conversation per Durable Object instance. Multi-conversation apps must: + +- Create separate DO instances per conversation (via `name` in `useAgent`) +- Manage conversation listing/metadata in the Worker entrypoint or a separate store +- Handle routing, access control, and lifecycle themselves + +There's no: + +- `listConversations(userId)` — users build this with KV/D1 +- Conversation metadata (title, created_at, last_message_at, message_count) +- Archive/delete conversation +- Client-side `useConversationList()` hook + +The examples handle this minimally — most use a single hardcoded conversation name or derive it from user ID. + +**Recommendation:** This is a feature gap, not an API bug. Consider providing: + +- A `ConversationManager` Worker helper that wraps DO instance lifecycle +- Metadata stored in the DO's SQLite (auto-maintained `last_message_at`, message count) +- A `/conversations` HTTP endpoint pattern +- A `useConversations()` client hook + +### Issue X3: No typed integration between chat and RPC + +`useAgent` provides typed RPC (`call`, `stub`) for invoking methods on the agent: + +```typescript +const agent = useAgent({ agent: "MyAgent", name: "room" }); +agent.call("setPreferences", { theme: "dark" }); // typed +``` + +But `useAgentChat` doesn't integrate with RPC. If an app needs both chat and custom methods (common for agents with configuration, side-panel actions, etc.), it must juggle both APIs: + +```typescript +const agent = useAgent({ agent: "MyAgent", name: "room" }); +const chat = useAgentChat({ agent }); + +// Chat via hook +chat.sendMessage(...); + +// Config via RPC — different API surface, same connection +await agent.call("setPreferences", { theme: "dark" }); +``` + +Think's `configure()` / `getConfig()` is a step toward typed configuration, but it's a separate mechanism from RPC. + +**Recommendation:** Consider a unified hook that exposes both chat and RPC surfaces. Or provide a `useAgentRPC(agent)` hook that shares the connection with `useAgentChat`. + +### Issue X4: Message rendering has no helpers + +The AI SDK's `UIMessage` uses a `parts`-based structure with many part types: + +```typescript +type UIMessage = { + id: string; + role: "user" | "assistant"; + parts: Array< + | { type: "text"; text: string } + | { type: "reasoning"; reasoning: string } + | { type: "tool-invocation"; toolCallId: string; toolName: string; state: string; ... } + | { type: "source"; source: { ... } } + | { type: "file"; data: string; mimeType: string } + | { type: "step-start" } + // ... + >; +}; +``` + +Every app must implement its own `parts.map(...)` renderer. Common needs that are reimplemented everywhere: + +- Text rendering with markdown +- Reasoning collapse/expand +- Tool state machine UI +- Source/citation rendering +- File/image display +- Step boundary visualization +- "Thinking..." placeholder for empty streaming messages + +**Recommendation:** Export a `` component with slot-based customization: + +```typescript + {part.text}} + renderTool={(part, helpers) => } + renderReasoning={(part) => {part.reasoning}} +/> +``` + +Or export utility hooks: `useMessageText(message)`, `useToolParts(message)`, `useReasoning(message)`. + +### Issue X5: Stream resume complexity + +The stream resume system works correctly but is complex across three layers: + +1. **Server**: `ResumableStream` buffers chunks in SQLite, `onConnect` sends `STREAM_RESUMING`, ACK handler replays chunks +2. **Transport**: `reconnectToStream` → `STREAM_RESUME_REQUEST` → wait for `STREAM_RESUMING` or `STREAM_RESUME_NONE` → `STREAM_RESUME_ACK` → `_createResumeStream` +3. **Hook**: `onAgentMessage` handles `STREAM_RESUMING` / `STREAM_RESUME_NONE` fallbacks, `broadcastTransition` processes replay chunks with `replay: true` / `replayComplete: true` flags, `localRequestIdsRef` prevents duplicate handling + +The tool continuation variant adds another code path (`expectToolContinuation` → `_createToolContinuationStream`), and the hook has a 5-second timeout for resume that silently resolves to `null`. + +This complexity is mostly internal and doesn't leak to users (the `resume: true` default handles everything). But it makes debugging resume issues very difficult, and the fallback path in `onAgentMessage` (lines 1218–1244) that handles `STREAM_RESUMING` when the transport isn't expecting it is particularly subtle. + +**Recommendation:** Consider consolidating resume into a single `ResumeManager` that encapsulates the state machine across transport and hook. Add debug logging (gated behind a flag) for resume lifecycle events. + +--- + +## Implications for Think + +> **Updated:** The analysis below was written before Session integration. See [think-roadmap.md](./think-roadmap.md) for the current plan. Key changes: +> +> - Session solves S3 (message access helpers — `getMessage`, `getLatestLeaf`, `getBranches`, `getPathLength`) +> - Session solves X2 (conversation management — `SessionManager` provides create/list/delete/rename/fork/search) +> - Context blocks provide structured system prompt composition — beyond what `getSystemPrompt()` alone offers +> - Compaction provides conversation length management — replacing `maxPersistedMessages` +> - Branching provides non-destructive regeneration — better than AIChatAgent's delete-and-rerun + +### What Think should adopt from AIChatAgent + +See [think-vs-aichat.md](./think-vs-aichat.md) for the full gap analysis. The high-priority items are: + +- `chatRecovery` / `onChatRecovery` (fiber-wrapped turns) +- `continueLastTurn()` (continuation as message append) +- `saveMessages()` (programmatic turn entry) +- `onChatResponse` (post-turn lifecycle hook) + +### What Think should NOT adopt + +1. **The `onFinish` parameter.** Think's `onChatMessage(options?)` is cleaner. If Think adds post-turn metadata, it should go through `onChatResponse` or a dedicated hook, not a callback parameter. + +2. **The `Response` return type.** Think's `StreamableResult` is the right abstraction. If backward compat with `Response` is needed, accept both via duck typing. + +3. **Deprecated-on-arrival options.** Think should not add `tools` (client tool schemas in hook), `toolsRequiringConfirmation`, or `experimental_automaticToolResolution`. The `onToolCall` + `needsApproval` pattern is the right API. + +4. **Complex concurrency without clear need.** Think uses queue-only concurrency. Adding `messageConcurrency` strategies should be demand-driven, not preemptive. + +5. **`maxPersistedMessages`.** Replaced by Session's compaction overlays — non-destructive summarization instead of lossy deletion. + +6. **`reconcileMessages`.** Session's tree structure with idempotent append and explicit `updateMessage` handles the core cases (ID conflicts, tool state merge) without a separate reconciliation pipeline. + +### Where Think can lead + +1. **Structured override points.** Think's `getModel()`, `getSystemPrompt()`, `getTools()`, `getMaxSteps()`, `configureSession()`, `assembleContext()` are much better than AIChatAgent's "override `onChatMessage` and wire everything yourself." These should be preserved and enhanced. + +2. **Sub-agent RPC (`chat()`).** AIChatAgent doesn't have this. The `chat(userMessage, callback, options?)` method for parent agents to drive sub-agent turns via RPC is a genuine differentiator for multi-agent architectures. This should be enhanced with `saveMessages`-equivalent programmatic turns. + +3. **Context blocks and compaction.** Session gives Think persistent, LLM-writable context blocks and non-destructive compaction — features AIChatAgent doesn't have. See [think-sessions.md](./think-sessions.md). + +4. **Branching and regeneration.** Session's tree-structured messages provide non-destructive regeneration (alternatives preserved via branches) — strictly better than AIChatAgent's delete-and-rerun approach. + +5. **Multi-session and search.** `SessionManager` provides conversation lifecycle, cross-session FTS5 search, usage tracking, and forking — none of which AIChatAgent has. + +6. **Dynamic configuration (`configure()` / `getConfig()`).** This is a cleaner pattern than `body` persistence for agent-level settings. It should coexist with `body` (per-request context) rather than replacing it. + +7. **Extension system.** `ExtensionManager` + sandboxed Worker tools are unique to Think and have no AIChatAgent equivalent. These should be preserved. + +8. **Client layer improvements.** The client-side issues (C1–C8) affect both Think and AIChatAgent equally since they share `useAgentChat`. Think could provide `useThinkChat` as an enhanced combined hook that simplifies setup and adds Think-specific features (config management, extension UI, context block display, branching navigation, session switching). + +--- + +## Summary: Prioritized Improvements + +### Tier 1: High impact, fix the pain + +| # | Issue | Impact | Effort | +| --- | ------------------------------------------------------- | ------------------------------------------------------------ | -------------- | +| C1 | Non-suspending hook + exported fetch + fallbackMessages | Unblocks framework integration, fixes conversation switching | Medium | +| S1 | Drop `onFinish` from `onChatMessage` signature | Cleaner API, less confusion | Low (breaking) | +| C3 | Tool UI components / utilities | Eliminates most repeated code across apps | Medium | +| C2 | Combined `useAgentChat` with built-in connection | Removes boilerplate for 90% of apps | Medium | + +### Tier 2: Significant quality-of-life + +| # | Issue | Impact | Effort | +| --- | ------------------------------------------------- | --------------------------------------------- | ------ | +| S2 | `continuation` flag on `OnChatMessageOptions` | Enables proper recovery/continuation handling | Low | +| C4 | Unified streaming status | Eliminates subtle streaming detection bugs | Low | +| S4 | Accept `StreamableResult \| Response` return type | Better abstraction, Think alignment | Medium | +| X4 | Message rendering helpers | Reduces boilerplate for every chat UI | Medium | +| C5 | Structured error handling on client | Proper error UX | Low | + +### Tier 3: Polish and completeness + +| # | Issue | Impact | Effort | +| --- | ---------------------------------------------- | ---------------------------------------- | -------------- | +| S3 | Message access helpers | Convenience, less raw array manipulation | Low | +| X1 | Remove deprecated options (major version) | Cleaner API surface, smaller bundle | Medium | +| C6 | `addToolOutput` naming alignment | Reduces confusion with AI SDK docs | Low (breaking) | +| X2 | Conversation management helpers | Supports multi-conversation apps | High | +| X3 | Chat + RPC integration | Unified agent interaction | Medium | +| C8 | Remove `@ts-expect-error` PartySocket coupling | Maintenance, reliability | Low | +| S5 | Document persistence timeline | Clarity for advanced users | Low | + +### Tier 4: Architectural evolution + +| # | Issue | Impact | Effort | +| --- | ------------------------------------- | ------------------------------ | ------ | +| X5 | Resume system consolidation | Debuggability, maintainability | High | +| C7 | Request cache improvements (SSR, TTL) | SSR correctness, cache hygiene | Medium | diff --git a/design/chat-improvements.md b/design/chat-improvements.md new file mode 100644 index 0000000000..1d83735646 --- /dev/null +++ b/design/chat-improvements.md @@ -0,0 +1,1040 @@ +# Chat Layer Improvements: Non-Breaking Changes + Shared Extraction + +> **Shared extraction (Wave 3) is complete.** `AbortRegistry`, `applyToolUpdate` + builders, `parseProtocolMessage`, and related primitives have been extracted into `agents/chat` and are consumed by both AIChatAgent and Think. See [think-roadmap.md](./think-roadmap.md) Phase 0 for details. Client-side improvements (Waves 1-2) and deprecation prep (Wave 4) remain as future work. + +Concrete improvements to `@cloudflare/ai-chat` (AIChatAgent + useAgentChat) that can ship without breaking changes, plus extraction of shared code into `agents/chat` to reduce duplication between AIChatAgent and Think. + +This document covers three concerns: + +1. **Non-breaking additions** to AIChatAgent and useAgentChat that improve DX immediately +2. **Shared code extraction** into `agents/chat` that benefits both AIChatAgent and Think (complete) +3. **Deprecation prep** for a future breaking change + +Related: + +- [chat-api.md](./chat-api.md) — API analysis identifying these issues +- [think-roadmap.md](./think-roadmap.md) — Think implementation plan (all phases complete) +- [think-sessions.md](./think-sessions.md) — Session integration design (implemented) + +--- + +## Table of Contents + +1. [Non-Breaking Additions](#non-breaking-additions) +2. [Shared Code Extraction](#shared-code-extraction) +3. [Deprecation Prep](#deprecation-prep) +4. [Implementation Order](#implementation-order) + +--- + +## Non-Breaking Additions + +### 1. Export `getAgentMessages()` from `@cloudflare/ai-chat` + +**Problem:** The `defaultGetInitialMessagesFetch` function in `react.tsx` is internal. Framework loaders (Next.js, TanStack Router, Remix) can't prefetch messages during route loading. Users must reconstruct the internal URL format manually. See [chat-api.md §C1](./chat-api.md#issue-c1-suspense-only-initial-message-fetch), [#1011](https://github.com/cloudflare/agents/issues/1011). + +**Solution:** Export a standalone fetch function from `@cloudflare/ai-chat/react` (or a separate `@cloudflare/ai-chat` entry): + +```typescript +export async function getAgentMessages< + M extends UIMessage = UIMessage +>(options: { + host: string; + agent: string; + name: string; + credentials?: RequestCredentials; + headers?: HeadersInit; +}): Promise { + const agentSlug = options.agent + .replace(/([a-z])([A-Z])/g, "$1-$2") + .toLowerCase(); + const url = new URL( + `${options.host}/agents/${agentSlug}/${options.name}/get-messages` + ); + const response = await fetch(url.toString(), { + credentials: options.credentials, + headers: options.headers + }); + + if (!response.ok) { + console.warn( + `Failed to fetch initial messages: ${response.status} ${response.statusText}` + ); + return []; + } + + const text = await response.text(); + if (!text.trim()) return []; + + try { + return JSON.parse(text) as M[]; + } catch { + console.warn("Failed to parse initial messages"); + return []; + } +} +``` + +**Usage in framework loaders:** + +```typescript +// TanStack Router +import { getAgentMessages } from "@cloudflare/ai-chat/react"; + +export const Route = createFileRoute("/chat/$conversationId")({ + loader: async ({ params }) => ({ + messages: await getAgentMessages({ + host: "https://my-app.workers.dev", + agent: "ChatAgent", + name: params.conversationId + }) + }) +}); + +// In component — skip Suspense, use loader data +const { messages: initialMessages } = Route.useLoaderData(); +const { messages, sendMessage } = useAgentChat({ + agent, + getInitialMessages: null, + messages: initialMessages +}); +``` + +**Effort:** Very low. The logic already exists internally — just export and document it. + +**Benefits Think:** Think speaks the same `/get-messages` protocol. The exported function works with Think agents too. + +### 2. Add `fallbackMessages` to `useAgentChat` + +**Problem:** Switching between conversations always suspends via `use()`, even if the user just visited that conversation. No stale-while-revalidate pattern. See [#1045](https://github.com/cloudflare/agents/issues/1045). + +**Solution:** Add an optional `fallbackMessages` option: + +```typescript +type UseAgentChatOptions = { + // ... existing options ... + + /** + * Messages to show immediately while the server fetch is in progress. + * Use for instant conversation switching with cached messages. + * + * When provided: + * 1. Messages render immediately (no Suspense) + * 2. Background fetch still runs via getInitialMessages + * 3. When fetch resolves, messages are replaced with server state + * 4. If user sends a message before fetch completes, fetch result is discarded + */ + fallbackMessages?: UIMessage[]; +}; +``` + +**Implementation sketch:** + +```typescript +// In useAgentChat: +const initialMessages = (() => { + if (options.fallbackMessages && initialMessagesPromise) { + // Don't suspend — use fallback immediately + return options.fallbackMessages; + } + if (initialMessagesPromise) { + return use(initialMessagesPromise); // existing Suspense path + } + return optionsInitialMessages ?? []; +})(); + +// Background revalidation effect (only when fallbackMessages is used) +useEffect(() => { + if (!options.fallbackMessages || !initialMessagesPromise) return; + + let stale = false; + initialMessagesPromise.then((serverMessages) => { + if (!stale && !hasSentMessage.current) { + setMessages(serverMessages); + } + }); + + return () => { + stale = true; + }; +}, [initialMessagesCacheKey]); +``` + +**Effort:** Low-medium. The fetch pipeline is unchanged — only the consumption side changes. + +**Benefits Think:** Same client hook, same benefit. + +### 3. Add `continuation` to `OnChatMessageOptions` + +**Problem:** `onChatMessage` doesn't know if it's being called for a continuation (after tool results, after recovery) vs a fresh user turn. Subclasses can't adjust system prompts, select different models, or skip expensive context assembly. See [chat-api.md §S2](./chat-api.md#issue-s2-onchatmessage-doesnt-know-if-its-a-continuation). + +**Solution:** Add to the existing type — purely additive: + +```typescript +export type OnChatMessageOptions = { + requestId: string; + abortSignal?: AbortSignal; + clientTools?: ClientToolSchema[]; + body?: Record; + /** True when this is a continuation (auto-continue after tool result, continueLastTurn, recovery). */ + continuation?: boolean; // NEW — additive, optional +}; +``` + +**Call sites to update (in AIChatAgent):** + +```typescript +// WebSocket user submit (line ~670) +{ requestId: chatMessageId, abortSignal, clientTools, body, continuation: false } + +// Auto-continuation after tool result (line ~1877) +{ requestId, abortSignal, clientTools, body, continuation: true } + +// continueLastTurn (line ~2177) +{ requestId, abortSignal, clientTools, body, continuation: true } + +// saveMessages / programmatic turn (line ~1948) +{ requestId, abortSignal, clientTools, body, continuation: false } +``` + +**Effort:** Very low. Add field to type, set at call sites. + +**Benefits Think:** Think's `ChatMessageOptions` should include the same field. Extract the type into `agents/chat` so both share it. + +### 4. Export tool part helpers + +**Problem:** Every app reimplements tool-state detection: `isToolUIPart(part)`, `getToolName(part)`, and the 8-state rendering switch. See [chat-api.md §C3](./chat-api.md#issue-c3-tool-ui-is-entirely-user-rebuilt-every-time). + +**Solution:** Export utilities from `@cloudflare/ai-chat/react`: + +```typescript +import type { UIMessage } from "ai"; + +type ToolUIPart = Extract< + UIMessage["parts"][number], + { type: "tool-invocation" } +>; + +/** + * Check if a message part is a tool invocation (any state). + */ +export function isToolUIPart( + part: UIMessage["parts"][number] +): part is ToolUIPart { + return part.type === "tool-invocation"; +} + +/** + * Get the tool name from a tool UI part. + */ +export function getToolName(part: ToolUIPart): string { + return (part as { toolName?: string }).toolName ?? "unknown"; +} + +/** + * Get a simplified state for rendering. + * Maps the 8+ internal states to 6 UI-relevant states. + */ +export function getToolPartState(part: ToolUIPart): + | "loading" // input-available, no approval needed + | "streaming" // input-streaming (partial input arriving) + | "waiting-approval" // approval-requested + | "approved" // approval-responded + | "complete" // output-available + | "error" // output-error + | "denied" { + // output-denied + const state = (part as { state?: string }).state; + switch (state) { + case "input-streaming": + return "streaming"; + case "approval-requested": + return "waiting-approval"; + case "approval-responded": + return "approved"; + case "output-available": + return "complete"; + case "output-error": + return "error"; + case "output-denied": + return "denied"; + default: + return "loading"; + } +} + +/** + * Get the tool output (if available). + */ +export function getToolOutput(part: ToolUIPart): unknown | undefined { + return (part as { output?: unknown }).output; +} + +/** + * Get the tool input (if available). + */ +export function getToolInput(part: ToolUIPart): unknown | undefined { + return (part as { input?: unknown }).input; +} + +/** + * Get the tool call ID. + */ +export function getToolCallId(part: ToolUIPart): string { + return (part as { toolCallId: string }).toolCallId; +} + +/** + * Get the approval info for a tool part (if in approval state). + */ +export function getToolApproval( + part: ToolUIPart +): { id: string; approved?: boolean } | undefined { + return (part as { approval?: { id: string; approved?: boolean } }).approval; +} +``` + +These are already used internally in `useAgentChat` (`isToolUIPart` at multiple call sites, `getToolName` throughout). Exporting them is a no-op change to internals. + +**Effort:** Very low. Extract existing internal functions, add `getToolPartState` and the accessor helpers. + +**Benefits Think:** Same exports work for Think agents since the message format is identical (`UIMessage` from AI SDK). + +### 5. Add `getHttpUrl()` to `useAgent` return value + +**Problem:** `useAgentChat` accesses internal `PartySocket` properties via `@ts-expect-error` to derive the HTTP URL for `/get-messages`. See [chat-api.md §C8](./chat-api.md#issue-c8-ts-expect-error-coupling-to-partysocket-internals). + +**Current (fragile):** + +```typescript +// packages/ai-chat/src/react.tsx, lines 469–474 +const agentUrl = new URL( + `${// @ts-expect-error we're using a protected _url property + ((agent._url as string | null) || agent._pkurl) + ?.replace("ws://", "http://") + .replace("wss://", "https://")}` +); +``` + +**Solution:** Add a public method to the `useAgent` return value (in `packages/agents/src/react.tsx`): + +```typescript +// In useAgent's return object: +const getHttpUrl = useCallback((): string => { + const wsUrl = socket._url || socket._pkurl; + return wsUrl?.replace("ws://", "http://").replace("wss://", "https://") ?? ""; +}, [socket]); + +return { + ...socket, + agent: resolvedAgent, + name: resolvedName, + // ... existing fields ... + getHttpUrl // NEW +}; +``` + +Then `useAgentChat` can replace the `@ts-expect-error` with: + +```typescript +const agentUrl = new URL(agent.getHttpUrl()); +``` + +**Effort:** Very low. Purely additive to `useAgent`'s return type. `useAgentChat` can migrate to it in the same PR. + +**Benefits Think:** Any Think-specific client hook would use the same public API instead of internal access. + +### 6. Add `isStreaming` property to AIChatAgent (server-side) + +**Problem:** AIChatAgent has no public way to check if a stream is active. `_resumableStream.hasActiveStream()` is private. Useful for HTTP endpoints (`GET /status`), RPC methods, and health checks. + +**Solution:** + +```typescript +// On AIChatAgent: +/** True when a chat stream is currently active. */ +get isStreaming(): boolean { + return this._resumableStream.hasActiveStream() || this._turnQueue.isActive; +} +``` + +**Effort:** Very low. + +**Benefits Think:** Same pattern applies to Think. Could extract into a shared mixin or just implement on both. + +### 7. Add `onChatError` callback to `useAgentChat` + +**Problem:** No structured error handling on the client for chat-level errors. Server errors arrive as `{ error: true }` frames but there's no callback to distinguish them from network errors. See [chat-api.md §C5](./chat-api.md#issue-c5-no-structured-error-handling). + +**Solution:** Add to `UseAgentChatOptions`: + +```typescript +type UseAgentChatOptions = { + // ... existing options ... + + /** + * Called when a chat error occurs. Receives the error message and source. + * Use for error toasts, logging, or custom error UI. + */ + onChatError?: (error: { + message: string; + requestId?: string; + source: "server" | "network" | "abort"; + }) => void; +}; +``` + +**Implementation:** In the `onAgentMessage` handler, when a `CF_AGENT_USE_CHAT_RESPONSE` has `error: true`: + +```typescript +if (data.error) { + onChatErrorRef.current?.({ + message: data.body || "Stream error", + requestId: data.id, + source: "server" + }); +} +``` + +In the transport's `sendMessages`, when `AbortError` is caught: + +```typescript +onChatErrorRef.current?.({ + message: "Request cancelled", + requestId, + source: "abort" +}); +``` + +**Effort:** Low. The error paths already exist — just need to call the callback. + +**Benefits Think:** Same callback works with Think agents. + +--- + +## Shared Code Extraction + +Code currently duplicated between AIChatAgent (`packages/ai-chat/src/index.ts`) and Think (`packages/think/src/think.ts`) that should be extracted into `agents/chat` (`packages/agents/src/chat/`). + +### 1. Protocol Handler Wiring + +**What's duplicated:** Both AIChatAgent and Think wrap `onConnect`, `onClose`, `onMessage`, and `onRequest` with the same pattern — intercept protocol messages, handle them, delegate to the user's original handler. + +**AIChatAgent** (lines 466–820 — ~354 lines of protocol handling in the constructor): + +```typescript +// packages/ai-chat/src/index.ts +const _onConnect = this.onConnect.bind(this); +this.onConnect = async (connection, ctx) => { + if (this._resumableStream.hasActiveStream()) { + this._notifyStreamResuming(connection); + } + return _onConnect(connection, ctx); +}; + +const _onClose = this.onClose.bind(this); +this.onClose = async (connection, code, reason, wasClean) => { + this._pendingResumeConnections.delete(connection.id); + this._continuation.awaitingConnections.delete(connection.id); + // ... continuation cleanup ... + return _onClose(connection, code, reason, wasClean); +}; + +const _onMessage = this.onMessage.bind(this); +this.onMessage = async (connection, message) => { + if (typeof message === "string") { + let data = JSON.parse(message); + // ... giant switch/if chain for all protocol messages ... + } + return _onMessage(connection, message); +}; +``` + +**Think** (lines 397–562 — ~165 lines with the same structure): + +```typescript +// packages/think/src/think.ts +private _setupProtocolHandlers() { + const _onConnect = this.onConnect.bind(this); + this.onConnect = async (connection, ctx) => { + if (this._resumableStream.hasActiveStream()) { + this._notifyStreamResuming(connection); + } + connection.send(JSON.stringify({ type: MSG_CHAT_MESSAGES, messages: this.messages })); + return _onConnect(connection, ctx); + }; + // ... same pattern for onClose, onMessage, onRequest ... +} +``` + +**Extraction:** Create a `ChatProtocolHandler` in `agents/chat` that encapsulates the hook wrapping pattern: + +```typescript +// agents/chat/protocol-handler.ts + +export interface ChatProtocolCallbacks { + // Protocol message handlers — each returns true if handled + onChatRequest( + connection: Connection, + data: Record + ): Promise; + onClear(connection: Connection): void; + onCancel(requestId: string): void; + onToolResult(connection: Connection, data: Record): void; + onToolApproval(connection: Connection, data: Record): void; + + // State queries + getMessages(): UIMessage[]; + hasActiveStream(): boolean; + getActiveRequestId(): string | null; + + // Stream resume + notifyStreamResuming(connection: Connection): void; + replayChunks(connection: Connection, requestId: string): string | null; + persistOrphanedStream(streamId: string): void; + + // Continuation state + getContinuation(): ContinuationState; + getPendingResumeConnections(): Set; +} + +export function setupChatProtocol( + agent: Agent, + callbacks: ChatProtocolCallbacks +): void { + // Wraps onConnect, onClose, onMessage, onRequest + // with protocol handling, delegating to callbacks for agent-specific behavior +} +``` + +Both AIChatAgent and Think would call `setupChatProtocol(this, { ... })` with their specific implementations, eliminating ~200+ lines of duplicated wrapping logic. + +**Effort:** Medium. Need to design the callback interface carefully to handle differences (AIChatAgent has `CF_AGENT_CHAT_MESSAGES` from client, concurrency decisions; Think doesn't). + +**Note:** An alternative is a simpler `parseProtocolMessage(data): { type, ...fields } | null` utility that both agents call inside their own `onMessage` handler. This is less ambitious but still eliminates the JSON parsing, type detection, and field extraction duplication. Think and AIChatAgent would still have their own switch/if chains, but each case body would be cleaner. + +### 2. Abort Controller Registry + +**What's duplicated:** Both maintain a `Map` with identical get/create/cancel/remove/destroy patterns. + +**AIChatAgent** (lines 3607–3645): + +```typescript +private _getAbortSignal(id: string): AbortSignal | undefined { + if (typeof id !== "string") return undefined; + if (!this._chatMessageAbortControllers.has(id)) { + this._chatMessageAbortControllers.set(id, new AbortController()); + } + return this._chatMessageAbortControllers.get(id)?.signal; +} + +private _removeAbortController(id: string) { + this._chatMessageAbortControllers.delete(id); +} + +private _cancelChatRequest(id: string) { + this._chatMessageAbortControllers.get(id)?.abort(); +} + +private _destroyAbortControllers() { + for (const controller of this._chatMessageAbortControllers.values()) { + controller?.abort(); + } + this._chatMessageAbortControllers.clear(); +} +``` + +**Think** (lines 607–608, 674–676, 703–708): + +```typescript +private _abortControllers = new Map(); + +// In _handleChatRequest: +const abortController = new AbortController(); +this._abortControllers.set(requestId, abortController); +// ... +this._abortControllers.delete(requestId); + +// In _handleClear: +for (const controller of this._abortControllers.values()) { + controller.abort(); +} +this._abortControllers.clear(); + +// In _handleCancel: +const controller = this._abortControllers.get(requestId); +if (controller) controller.abort(); +``` + +**Extraction:** + +```typescript +// agents/chat/abort-registry.ts + +export class AbortRegistry { + private controllers = new Map(); + + /** Get or create an AbortController for the given ID. Returns its signal. */ + getSignal(id: string): AbortSignal { + if (!this.controllers.has(id)) { + this.controllers.set(id, new AbortController()); + } + return this.controllers.get(id)!.signal; + } + + /** Cancel a specific request. */ + cancel(id: string): void { + this.controllers.get(id)?.abort(); + } + + /** Remove a controller after the request completes. */ + remove(id: string): void { + this.controllers.delete(id); + } + + /** Abort all pending requests and clear. */ + destroyAll(): void { + for (const controller of this.controllers.values()) { + controller.abort(); + } + this.controllers.clear(); + } + + /** Check if a request is tracked. */ + has(id: string): boolean { + return this.controllers.has(id); + } +} +``` + +**Effort:** Very low. Self-contained utility, no interface design needed. + +### 3. Tool State Machine + +**What's duplicated:** Both implement `_applyToolResult` and `_applyToolApproval` with the same state matching logic — find the message containing a tool part with the given `toolCallId`, check it's in a valid state, apply the update, persist, broadcast. + +**AIChatAgent** (lines 2809–2959 — ~150 lines): + +- `_findAndUpdateToolPart()` — generic find-and-update with retry backoff for streaming race +- `_applyToolResult()` — delegates to `_findAndUpdateToolPart` with result-specific update +- `_applyToolApproval()` — delegates to `_findAndUpdateToolPart` with approval-specific update + +**Think** (lines 930–1001 — ~70 lines): + +- `_applyToolResult()` — inline find loop, update, persist, broadcast +- `_applyToolApproval()` — inline find loop, update, persist, broadcast + +The core logic is the same — iterate message parts, find matching `toolCallId` in valid states, apply update. The differences are: + +- AIChatAgent has a retry loop for streaming race conditions (`_findAndUpdateToolPart` with backoff) +- AIChatAgent separates streaming vs persisted message paths (in-place mutation vs immutable update) +- Think always operates on persisted messages + +**Extraction:** Extract the state matching and update logic as a pure function: + +```typescript +// agents/chat/tool-state.ts + +export type ToolPartUpdate = { + toolCallId: string; + matchStates: string[]; + apply: (part: Record) => Record; +}; + +/** + * Find and update a tool part in a message array. + * Returns the updated message (or null if not found), and the part index. + */ +export function findAndUpdateToolPart( + messages: UIMessage[], + update: ToolPartUpdate +): { message: UIMessage; partIndex: number; updated: UIMessage } | null { + for (const msg of messages) { + for (let i = 0; i < msg.parts.length; i++) { + const part = msg.parts[i] as Record; + if ( + "toolCallId" in part && + part.toolCallId === update.toolCallId && + "state" in part && + update.matchStates.includes(part.state as string) + ) { + const updatedParts = [...msg.parts]; + updatedParts[i] = update.apply(part) as UIMessage["parts"][number]; + return { + message: msg, + partIndex: i, + updated: { ...msg, parts: updatedParts } as UIMessage + }; + } + } + } + return null; +} + +/** Pre-built update for tool result application. */ +export function toolResultUpdate( + toolCallId: string, + output: unknown, + overrideState?: "output-error", + errorText?: string +): ToolPartUpdate { + return { + toolCallId, + matchStates: [ + "input-available", + "approval-requested", + "approval-responded" + ], + apply: (part) => ({ + ...part, + ...(overrideState === "output-error" + ? { + state: "output-error", + errorText: errorText ?? "Tool execution denied by user" + } + : { state: "output-available", output, preliminary: false }) + }) + }; +} + +/** Pre-built update for tool approval application. */ +export function toolApprovalUpdate( + toolCallId: string, + approved: boolean +): ToolPartUpdate { + return { + toolCallId, + matchStates: ["input-available", "approval-requested"], + apply: (part) => ({ + ...part, + state: approved ? "approval-responded" : "output-denied", + approval: { + ...(part.approval as Record | undefined), + approved + } + }) + }; +} +``` + +Both agents would call `findAndUpdateToolPart(this.messages, toolResultUpdate(toolCallId, output))` and then handle persistence and broadcast in their own way. AIChatAgent keeps its retry/streaming logic around the call; Think keeps its simpler inline path. + +**Effort:** Low-medium. The pure function extraction is straightforward. The tricky part is AIChatAgent's streaming message path (`_streamingMessage` in-place mutation) — that stays in AIChatAgent, but the state matching and update construction moves to the shared function. + +### 4. Request Context Persistence + +**What's duplicated:** Both persist key-value context (client tools, body) to SQLite with identical patterns. + +**AIChatAgent** (lines 1149–1193): + +```typescript +private _restoreRequestContext() { + const rows = this.sql`select key, value from cf_ai_chat_request_context` || []; + for (const row of rows) { + if (row.key === "lastBody") this._lastBody = JSON.parse(row.value); + else if (row.key === "lastClientTools") this._lastClientTools = JSON.parse(row.value); + } +} + +private _persistRequestContext() { + if (this._lastBody) { + this.sql`insert or replace into cf_ai_chat_request_context (key, value) + values ('lastBody', ${JSON.stringify(this._lastBody)})`; + } else { + this.sql`delete from cf_ai_chat_request_context where key = 'lastBody'`; + } + // ... same for lastClientTools ... +} +``` + +**Think** (lines 865–888): + +```typescript +private _persistClientTools(): void { + if (this._lastClientTools) { + this.sql`INSERT OR REPLACE INTO think_request_context (key, value) + VALUES ('lastClientTools', ${JSON.stringify(this._lastClientTools)})`; + } else { + this.sql`DELETE FROM think_request_context WHERE key = 'lastClientTools'`; + } +} + +private _restoreClientTools(): void { + const rows = this.sql`SELECT value FROM think_request_context WHERE key = 'lastClientTools'` || []; + if (rows.length > 0) { + this._lastClientTools = JSON.parse(rows[0].value); + } +} +``` + +Different table names, same pattern. + +**Extraction:** + +```typescript +// agents/chat/request-context.ts + +export interface SqlProvider { + sql>( + strings: TemplateStringsArray, + ...values: (string | number | boolean | null)[] + ): T[]; +} + +export class RequestContextStore { + private agent: SqlProvider; + private table: string; + private initialized = false; + + constructor(agent: SqlProvider, table = "cf_chat_request_context") { + this.agent = agent; + this.table = table; + } + + private ensureTable(): void { + if (this.initialized) return; + this.agent.sql`CREATE TABLE IF NOT EXISTS ${this.table} ( + key TEXT PRIMARY KEY, value TEXT NOT NULL + )`; + this.initialized = true; + } + + get(key: string): T | undefined { + this.ensureTable(); + const rows = this.agent.sql<{ value: string }>` + SELECT value FROM ${this.table} WHERE key = ${key} + `; + if (rows.length === 0) return undefined; + try { + return JSON.parse(rows[0].value) as T; + } catch { + return undefined; + } + } + + set(key: string, value: unknown): void { + this.ensureTable(); + if (value !== undefined && value !== null) { + this.agent.sql`INSERT OR REPLACE INTO ${this.table} (key, value) + VALUES (${key}, ${JSON.stringify(value)})`; + } else { + this.agent.sql`DELETE FROM ${this.table} WHERE key = ${key}`; + } + } + + getAll(): Record { + this.ensureTable(); + const rows = + this.agent.sql<{ key: string; value: string }>` + SELECT key, value FROM ${this.table} + ` || []; + const result: Record = {}; + for (const row of rows) { + try { + result[row.key] = JSON.parse(row.value); + } catch { + /* skip corrupted */ + } + } + return result; + } + + clear(): void { + this.ensureTable(); + this.agent.sql`DELETE FROM ${this.table}`; + } +} +``` + +**Note:** For Think-on-Session, this functionality moves to `assistant_config` — but the `RequestContextStore` interface is still useful for AIChatAgent (which doesn't use Session). The store could accept `assistant_config` as the table name when Session is in play. + +**Effort:** Low. Self-contained utility. + +### 5. Stream Resume Handshake + +**What's duplicated:** Both implement the `_notifyStreamResuming` → `STREAM_RESUME_REQUEST` → `STREAM_RESUME_ACK` → `replayChunks` → `_persistOrphanedStream` pattern with nearly identical logic. + +**AIChatAgent** (lines 760–820 for the ACK handler, lines 989–1005 for `_notifyStreamResuming`): + +```typescript +private _notifyStreamResuming(connection: Connection) { + if (!this._resumableStream.hasActiveStream()) return; + this._pendingResumeConnections.add(connection.id); + connection.send(JSON.stringify({ + type: MessageType.CF_AGENT_STREAM_RESUMING, + id: this._resumableStream.activeRequestId + })); +} + +// ACK handler: +if (data.type === MessageType.CF_AGENT_STREAM_RESUME_ACK) { + this._pendingResumeConnections.delete(connection.id); + if (this._resumableStream.hasActiveStream() && + this._resumableStream.activeRequestId === data.id) { + const orphanedStreamId = this._resumableStream.replayChunks(connection, ...); + if (orphanedStreamId) this._persistOrphanedStream(orphanedStreamId); + } + return; +} +``` + +**Think** (lines 465–501 for the handler, lines 1104–1113 for `_notifyStreamResuming`): +Identical logic with different message constant names (`MSG_STREAM_RESUME_ACK` vs `MessageType.CF_AGENT_STREAM_RESUME_ACK`). + +**Extraction:** Since `ResumableStream` already lives in `agents/chat`, extend it with resume handshake methods: + +```typescript +// Add to ResumableStream or a new StreamResumeHandler: + +export class StreamResumeHandler { + private stream: ResumableStream; + private pendingConnections: Set; + private continuation: ContinuationState; + + notifyStreamResuming(connection: Connection): void { ... } + + handleResumeRequest( + connection: Connection, + continuation: ContinuationState + ): "resuming" | "awaiting-continuation" | "none" { ... } + + handleResumeAck( + connection: Connection, + requestId: string, + persistOrphan: (streamId: string) => void + ): void { ... } +} +``` + +**Effort:** Medium. The continuation state interactions make this slightly complex. + +### 6. Broadcast with Resume Exclusions + +**What's duplicated:** Both exclude `_pendingResumeConnections` from broadcasts. + +**AIChatAgent** (lines 1195–1204): + +```typescript +private _broadcastChatMessage(message: OutgoingMessage, exclude?: string[]) { + const allExclusions = [...(exclude || []), ...this._pendingResumeConnections]; + this.broadcast(JSON.stringify(message), allExclusions); +} +``` + +**Think** (lines 1131–1147 — `_broadcastChat` method): + +```typescript +private _broadcastChat(payload: Record, exclude?: string[]): void { + const allExclude = exclude + ? [...exclude, ...this._pendingResumeConnections] + : [...this._pendingResumeConnections]; + this.broadcast(JSON.stringify(payload), allExclude); +} +``` + +This is minor duplication (~6 lines each) but conceptually belongs with the resume handler. If `StreamResumeHandler` owns `_pendingResumeConnections`, it can provide a `getExclusions()` method. + +**Effort:** Very low (part of resume handler extraction). + +--- + +## Deprecation Prep + +These don't change behavior — they add warnings and documentation that prepare users for a future breaking change. + +### 1. Deprecate `onFinish` parameter on `onChatMessage` + +**What:** Add `@deprecated` JSDoc. Add a one-time runtime warning if a subclass's `onChatMessage` override actually uses the `_finishResult` parameter (detect by checking if the passed callback was called with non-empty data). + +**Migration path:** Use `onChatResponse` for post-turn metadata. + +**Think:** Never adds `onFinish`. Think's `onChatMessage(options?)` is the target signature. + +### 2. Deprecate `addToolOutput` naming + +**What:** Add `addToolResult` as an alias in the return value of `useAgentChat`. Deprecate `addToolOutput` in JSDoc. Both call the same internal function. + +```typescript +return { + // ... existing return ... + addToolOutput, // @deprecated — use addToolResult + addToolResult: addToolOutput // NEW alias +}; +``` + +**Migration path:** Switch from `addToolOutput` to `addToolResult`. + +### 3. Deprecate remaining legacy options + +**Already deprecated but not yet removed:** + +- `tools` (with `execute`) → use `onToolCall` +- `experimental_automaticToolResolution` → use `onToolCall` +- `toolsRequiringConfirmation` → use `needsApproval` on server tools +- `autoSendAfterAllConfirmationsResolved` → use `sendAutomaticallyWhen` from AI SDK + +**Action:** Add `@deprecated` JSDoc if missing. Add deprecation notice to README. Plan removal in next major version. Consider moving deprecated code paths to `@cloudflare/ai-chat/compat`. + +### 4. Mark `Response` return type for future change + +**What:** Document in `onChatMessage` JSDoc that a future version will accept `StreamableResult | Response` (or just `StreamableResult`). Don't change the type yet — just signal the direction. + +--- + +## Implementation Order + +### Wave 1: Quick wins (1–2 days each) + +These are independent, purely additive, and can be PRed in parallel: + +| # | Change | Effort | Impact | +| --- | -------------------------------------------- | -------- | ----------------------------------- | +| 1 | Export `getAgentMessages()` | Very low | Unblocks framework loaders | +| 3 | Add `continuation` to `OnChatMessageOptions` | Very low | Better continuation handling | +| 4 | Export tool part helpers | Very low | Eliminates boilerplate in every app | +| 5 | Add `getHttpUrl()` to `useAgent` | Very low | Removes `@ts-expect-error` | +| 6 | Add `isStreaming` to AIChatAgent | Very low | Server-side stream detection | + +### Wave 2: Client DX (3–5 days) + +| # | Change | Effort | Impact | +| --- | ---------------------------------------- | ---------- | ---------------------------- | +| 2 | Add `fallbackMessages` to `useAgentChat` | Low-medium | Fixes conversation switching | +| 7 | Add `onChatError` to `useAgentChat` | Low | Structured error handling | + +### Wave 3: Extraction (benefits Think Phase 1) + +These should land before Think's Session integration PR, so Think can import from `agents/chat`: + +| # | Change | Effort | Impact | +| --- | ---------------------------------------------------- | ---------- | ----------------------------------------- | +| E2 | Extract `AbortRegistry` | Very low | Think reuses, AIChatAgent simplified | +| E4 | Extract `RequestContextStore` | Low | Think reuses (or uses `assistant_config`) | +| E3 | Extract tool state machine (`findAndUpdateToolPart`) | Low-medium | Think reuses, AIChatAgent simplified | +| E1 | Extract protocol handler (or `parseProtocolMessage`) | Medium | Largest dedup win | +| E5 | Extract stream resume handler | Medium | Think reuses, resume logic consolidated | + +### Wave 4: Deprecation + +Can happen anytime, independent of the above: + +| # | Change | Effort | Notes | +| --- | ----------------------------------- | -------- | --------------------------------- | +| D1 | Deprecate `onFinish` | Very low | JSDoc + optional runtime warning | +| D2 | Add `addToolResult` alias | Very low | Alias + deprecate `addToolOutput` | +| D3 | Deprecate legacy hook options | Very low | JSDoc updates | +| D4 | Signal `StreamableResult` direction | Very low | JSDoc only | + +### Dependency on Think roadmap + +Wave 3 extractions directly feed into [Think Phase 1](./think-roadmap.md#phase-1-session-integration). The ideal timeline: + +``` +Wave 1 (quick wins) ─┐ +Wave 2 (client DX) ─┤─ Can proceed in parallel +Wave 4 (deprecation) ─┘ + │ +Wave 3 (extraction) ───┤─ Land before Think Phase 1 + │ +Think Phase 1 ───┘─ Session integration, imports from agents/chat +``` diff --git a/design/chat-shared-layer.md b/design/chat-shared-layer.md new file mode 100644 index 0000000000..0d9ab671ca --- /dev/null +++ b/design/chat-shared-layer.md @@ -0,0 +1,279 @@ +# Chat Shared Layer + +Shared streaming, persistence, and protocol primitives for the `cf_agent_chat_*` WebSocket protocol. Lives in `packages/agents/src/chat/` and is consumed by both `@cloudflare/ai-chat` (the stable chat agent) and `@cloudflare/think` (the opinionated assistant base class). + +## Problem + +`@cloudflare/ai-chat` and `@cloudflare/think` both implement the same WebSocket chat protocol and share fundamental streaming/persistence concerns, but they live in separate packages with no shared code path. Think was forced to **fork** `message-builder.ts` (with a drift warning comment) and reimplement sanitization because `agents` — the only package both depend on — didn't have these primitives. + +This led to: + +- **Duplicated chunk-to-message logic** (`applyChunkToParts`) across two packages, with a comment warning about drift risk +- **Duplicated sanitization** (OpenAI metadata stripping, row-size enforcement) with subtle behavioral differences +- **Duplicated wire protocol constants** (`MSG_CHAT_*` strings matching `MessageType` values) +- **Duplicated metadata handling** (the `start`/`finish`/`message-metadata` switch that `applyChunkToParts` doesn't cover) in three separate code paths: ai-chat server, ai-chat client, and Think server + +On the ai-chat side, `index.ts` (~3700 lines) and `react.tsx` (~1577 lines) mixed too many concerns together — streaming, reconciliation, persistence, broadcasting, turn management — making the code difficult to modify and reason about. + +## Architecture + +``` +packages/agents/src/chat/ ← shared foundation + index.ts barrel exports + message-builder.ts applyChunkToParts + types + sanitize.ts sanitizeMessage, enforceRowSizeLimit + stream-accumulator.ts StreamAccumulator class + turn-queue.ts TurnQueue class + broadcast-state.ts broadcastTransition state machine + resumable-stream.ts ResumableStream (SQLite chunk buffer) + client-tools.ts ClientToolSchema, createToolsFromClientSchemas + protocol.ts CHAT_MESSAGE_TYPES constants (chat + resume + tool) + +packages/ai-chat/src/ ← stable chat agent + client + index.ts AIChatAgent (uses shared imports) + react.tsx useAgentChat (uses broadcastTransition) + message-reconciler.ts reconcileMessages, resolveToolMergeId + ws-chat-transport.ts WebSocket transport for AI SDK + types.ts MessageType enum, wire protocol types + +packages/think/src/ ← opinionated assistant + think.ts Think (uses shared imports) + extensions/ ExtensionManager, HostBridgeLoopback (standalone) +``` + +**Dependency direction**: `ai-chat → agents`, `think → agents`. The shared layer resolves the circular dependency that caused the original fork. + +## Modules + +### message-builder.ts + +**`applyChunkToParts(parts, chunk) → boolean`** — the core chunk-to-message-part builder. Mutates a `UIMessage["parts"]` array in place for streaming performance. Returns `true` if the chunk type was recognized, `false` for types the caller must handle (`start`, `finish`, `message-metadata`, `error`, `finish-step`). + +This is the single most shared piece of code in the chat system. Used by: + +- `AIChatAgent._streamSSEReply` — server-side SSE parsing +- `AIChatAgent._persistOrphanedStream` — rebuilding messages from stored chunks after hibernation +- `StreamAccumulator.applyChunk` — the higher-level wrapper +- Think's `StreamAccumulator` usage in `_streamResult` and `chat()` + +**Key type: `StreamChunkData`** — deliberately loose (index signature, many optionals) to match the wire format without encoding chunk-type-specific constraints. The `messageMetadata` field is typed as `unknown` (not `Record`) to match `UIMessageChunk` from the AI SDK. + +### sanitize.ts + +Two functions for persistence hygiene: + +**`sanitizeMessage(message) → UIMessage`** — strips OpenAI ephemeral fields (`itemId`, `reasoningEncryptedContent`) from `providerMetadata` and `callProviderMetadata`, then filters truly empty reasoning parts (no text and no remaining provider metadata after stripping). + +**`enforceRowSizeLimit(message) → UIMessage`** — compacts messages exceeding 1.8MB (the safety threshold below SQLite's 2MB row limit). Two-pass: first compact tool outputs over 1KB, then truncate text parts. + +`@cloudflare/ai-chat` wraps these with additional logic: + +- `_truncateProviderExecutedToolPayloads` — truncates large strings in Anthropic-style server-executed tool payloads (code_execution, text_editor) +- `sanitizeMessageForPersistence()` — protected hook for subclass customization +- `_enforceRowSizeLimit` — adds `console.warn` logging and `metadata.compactedToolOutputs` / `metadata.compactedTextParts` tracking + +Think uses the shared functions directly (no extra steps). + +### stream-accumulator.ts + +**`StreamAccumulator`** — wraps `applyChunkToParts` and handles the chunk types it returns `false` for. Manages `messageId`, `parts`, and `metadata` as a coherent unit. + +```typescript +class StreamAccumulator { + messageId: string; + readonly parts: UIMessage["parts"]; + metadata?: Record; + + applyChunk(chunk: StreamChunkData): ChunkResult; + toMessage(): UIMessage; + mergeInto(messages: UIMessage[]): UIMessage[]; +} +``` + +**`ChunkResult`** carries an optional **`ChunkAction`** — a discriminated union that signals domain-specific concerns without the accumulator knowing about them: + +| Action type | When | Caller handles | +| --------------------------- | ------------------------------------------------------------------------------------- | ----------------------------------------------------------------------- | +| `start` | `start` chunk with optional `messageId` / `messageMetadata` | ai-chat: may overwrite `message.id` | +| `finish` | `finish` chunk with optional `finishReason` | ai-chat: normalize `finishReason` to `messageMetadata` before broadcast | +| `message-metadata` | `message-metadata` chunk | Metadata already merged by accumulator | +| `tool-approval-request` | `tool-approval-request` chunk | ai-chat: early persist to SQLite for page-refresh survival | +| `cross-message-tool-update` | `tool-output-available` / `tool-output-error` for a `toolCallId` not in current parts | ai-chat: search `this.messages` and update persisted message | +| `error` | `error` chunk | Think: broadcast error frame, `continue`; ai-chat: broadcast error | + +**`mergeInto(messages)`** — produces a new message array by finding an existing message (by `messageId`, or walking backward for last assistant in continuation mode), then replacing or appending. This replaced the `flushActiveStreamToMessages` function on the client and the `activeStreamRef` + metadata merge pattern. + +**Where the accumulator is used vs. not:** + +- **ai-chat client** (`react.tsx`): Uses `StreamAccumulator` for broadcast/resume streams. The transport-owned path (local tab requests) still goes through `useChat`'s built-in pipeline. +- **Think server**: Uses `StreamAccumulator` in both `_streamResult` (WebSocket path) and `chat()` (RPC sub-agent path). +- **ai-chat server** (`_streamSSEReply`): Still uses `applyChunkToParts` directly. The server's streaming message (`_streamingMessage`) is shared by reference with `hasPendingInteraction`, `_messagesForClientSync`, and `_findAndUpdateToolPart`, making it impractical to route through the accumulator without a deeper refactoring of the shared mutable state. + +### protocol.ts + +**`CHAT_MESSAGE_TYPES`** — plain string constants for the wire protocol message types. Used by Think to avoid depending on `@cloudflare/ai-chat/types` (which would create a dependency edge Think shouldn't have). The values match `MessageType` in `ai-chat/src/types.ts`. + +### message-reconciler.ts (ai-chat only) + +Pure functions for aligning client messages with server state during persistence. Think doesn't need these — its `INSERT OR IGNORE` + reload-from-DB model avoids the ID reconciliation problem entirely. + +**`reconcileMessages(incoming, serverMessages, sanitize?)`** — two-stage pipeline: + +1. **Tool output merge**: When the server has `output-available` for a tool that the client still shows as `input-available`, `approval-requested`, or `approval-responded`, adopt the server's output. This handles the case where the client sends stale tool states. + +2. **ID reconciliation** (two-pass): + - Pass 1: Exact ID matches between incoming and server, claiming server indices + - Pass 2: Content-key matching for non-tool assistant messages using JSON-serialized sanitized parts. Prevents duplicate rows when the AI SDK assigns a different local ID than the server. + +**`resolveToolMergeId(message, serverMessages)`** — per-message ID resolution by `toolCallId`. If a tool call ID exists in a server message with a different ID, adopt the server's ID. Called during persistence to prevent duplicate rows. + +## Key decisions + +### Why `agents/chat` and not a new package + +Both `ai-chat` and `think` already depend on `agents`. Adding a new package would create another dependency edge and another build/publish step. The `agents` package already has subdirectory exports (`agents/mcp`, `agents/react`, etc.), so `agents/chat` follows the established pattern. + +### Why the accumulator signals actions instead of handling them + +The accumulator doesn't know about SQLite, WebSockets, or broadcasting. It signals via `ChunkAction` and the caller decides what to do. This keeps the accumulator testable as a pure data structure and reusable across contexts that handle actions differently (server persists to SQLite on approval, client ignores it; server broadcasts errors on the wire, client logs them). + +### Why `_streamSSEReply` was not refactored to use the accumulator + +`_streamSSEReply` in ai-chat's server mutates a `message` object that is shared by reference as `this._streamingMessage`. Other methods read this reference to check for pending tool interactions, build client sync payloads, and apply tool results during streaming. Routing through a `StreamAccumulator` would require either: + +1. Sharing the same parts array between the accumulator and the message object (breaking the accumulator's encapsulation) +2. Syncing the accumulator's state back to the message after each chunk (adding complexity, not removing it) +3. Refactoring all consumers of `_streamingMessage` to read from the accumulator (a much larger change) + +None of these reduce complexity. The metadata handling on the server is ~30 lines of straightforward switch/case that matches the accumulator's behavior exactly. The cost of duplication is low; the risk of the refactoring is high. + +### Why reconciliation stays in ai-chat + +Think avoids the reconciliation problem entirely through its persistence model: user messages use `INSERT OR IGNORE` (idempotent), assistant messages use `INSERT ON CONFLICT UPDATE`, and the authoritative message list is always reloaded from SQLite. There's no client/server ID mismatch because Think controls the full lifecycle. + +`AIChatAgent` can't do this because it must accept whatever IDs the AI SDK generates on the client side, and the `useChat` hook's internal state management can produce ID mismatches during streaming, tool interactions, and page refreshes. + +### Why `StreamChunkData.messageMetadata` is `unknown` + +The AI SDK's `UIMessageChunk` types `messageMetadata` as `unknown`. If `StreamChunkData` used `Record`, passing a `UIMessageChunk` directly to `applyChunkToParts` would fail type checking. The accumulator uses an `asMetadata()` helper to safely narrow `unknown` to `Record` at runtime. + +## Tradeoffs + +**Shared `enforceRowSizeLimit` lacks ai-chat's observability features.** The shared version doesn't add `metadata.compactedToolOutputs` or `console.warn` on compaction. Think gets the simpler version; ai-chat wraps it with its own enhanced version. If Think ever needs compaction observability, the shared function could accept an options bag. + +**The accumulator creates a new message on every `toMessage()` / `mergeInto()` call.** This is intentional for immutability (React needs new references for re-renders), but it means the server can't use `toMessage()` for its shared `_streamingMessage` reference without breaking identity. + +**Wire protocol constants are duplicated between `CHAT_MESSAGE_TYPES` and `MessageType`.** The values are identical strings but live in two places. `MessageType` is `@cloudflare/ai-chat`'s published enum; `CHAT_MESSAGE_TYPES` is `agents`'s internal constants. Drift is the operational risk. A future consolidation could move the canonical values to `agents/chat` and have `ai-chat` re-export them, but that requires `ai-chat` to depend on the specific export path — a semver-sensitive change. + +## What's next + +### TurnQueue (done) + +`TurnQueue` — a serial async queue with generation-based invalidation — now lives in `agents/chat/turn-queue.ts`. It handles: + +- Promise-chain serialization (FIFO) +- Generation counter with `reset()` (maps to ai-chat's epoch and Think's former `_clearGeneration`) +- Auto-skip of stale entries (generation mismatch at the front of the queue) +- Active request tracking (`activeRequestId`, `isActive`) +- `waitForIdle()` — resolves when the queue is fully drained +- Per-generation queued counts (`queuedCount()`) + +```typescript +class TurnQueue { + get generation(): number; + get activeRequestId(): string | null; + get isActive(): boolean; + enqueue( + requestId: string, + fn: () => Promise, + options?: EnqueueOptions + ): Promise>; + reset(): void; + waitForIdle(): Promise; + queuedCount(generation?: number): number; +} +``` + +**AIChatAgent** uses it through `_runExclusiveChatTurn`, which wraps `_turnQueue.enqueue()` with the `onChatResponse` drain and merge-map cleanup. Concurrency policies (drop/latest/merge/debounce) remain in AIChatAgent — they operate on message-specific state the queue doesn't know about. The `onStale` callback on `_runExclusiveChatTurn` lets the WS submit call site send a `done:true` response for turns skipped by auto-skip. + +**Think** wraps both `chat()` and `_handleChatRequest` in `_turnQueue.enqueue()`, giving it proper turn serialization (previously concurrent calls could interleave on `this.messages`). `_clearGeneration` was replaced by `_turnQueue.generation`. + +**Fields moved from AIChatAgent to TurnQueue:** `_chatTurnQueue`, `_activeChatTurnRequestId`, `_chatEpoch`, `_queuedChatTurnCountsByEpoch`. + +**Fields that stayed in AIChatAgent:** `_mergeQueuedUserStartIndexByEpoch`, `_submitSequence` / `_latestOverlappingSubmitSequence`, `_activeDebounceTimer` / `_activeDebounceResolve`, `_pendingChatResponseResults` / `_insideResponseHook`, `_pendingInteractionPromise`. + +--- + +### Server-side StreamAccumulator (deferred) + +Making `_streamSSEReply` use the `StreamAccumulator` requires resolving the `_streamingMessage` shared reference problem. + +**Consumers of `_streamingMessage`:** + +| Method | What it reads | Mutation? | +| ------------------------ | ------------------------------------------------------------------- | ---------------------------------------------------- | +| `_messagesForClientSync` | `parts.length`, `id`, full object (spliced into messages array) | Read only | +| `hasPendingInteraction` | Full object → `_messageHasPendingInteraction` | Read only | +| `_findAndUpdateToolPart` | Iterates `parts`, uses `message === _streamingMessage` for identity | **Mutates parts in place** when `isStreamingMessage` | +| `_streamSSEReply` | Truthiness, shallow copy for early persist snapshot | Read only | +| `_reply` | Sets to the live message object; clears to `null` in `finally` | Write | + +**The core problem:** `_findAndUpdateToolPart` uses **reference identity** (`message === this._streamingMessage`) to decide whether to mutate parts in place vs. spread-copy. If `_streamingMessage` were a `StreamAccumulator`, you'd need to replace this identity check with something else (e.g., a boolean `isStreamingBuffer`, or comparing against the accumulator instance). + +**Possible approaches:** + +1. **Shared parts array.** Make `StreamAccumulator` accept an external `parts` array in its constructor (by reference, not copy). The accumulator and the `ChatMessage` share the same array. `applyChunk` mutates the shared array. `_streamingMessage` continues to point to the `ChatMessage`. The accumulator is only used for metadata handling. **Downside:** Breaks the accumulator's current encapsulation (constructor copies parts). + +2. **Accumulator as `_streamingMessage`.** Replace `_streamingMessage: ChatMessage | null` with `_streamingAccumulator: StreamAccumulator | null`. Refactor all consumers to use `_streamingAccumulator.parts` / `_streamingAccumulator.messageId` / `_streamingAccumulator.toMessage()`. The biggest change is `_findAndUpdateToolPart`'s identity check — replace with `message === _streamingAccumulator?.toMessage()` won't work (toMessage creates new objects). Use a flag instead. **Downside:** Touches 5+ methods. + +3. **Leave as-is.** The metadata handling in `_streamSSEReply` is ~30 lines of switch/case that exactly matches the accumulator's behavior. The cost of duplication is low. **This is the current state.** + +--- + +### Broadcast stream state machine (done) + +`broadcastTransition` — a pure state machine for the accumulator-based broadcast/resume path — now lives in `agents/chat/broadcast-state.ts`. It manages the `StreamAccumulator` lifecycle that `useAgentChat`'s `onAgentMessage` handler previously tracked through scattered refs (`accumulatorRef`, `activeStreamIdRef`). + +```typescript +type BroadcastStreamState = + | { status: "idle" } + | { status: "observing"; streamId: string; accumulator: StreamAccumulator }; + +type BroadcastStreamEvent = + | { + type: "response"; + streamId: string; + messageId: string; + chunkData?: unknown; + done?: boolean; + error?: boolean; + replay?: boolean; + replayComplete?: boolean; + continuation?: boolean; + currentMessages?: UIMessage[]; + } + | { type: "resume-fallback"; streamId: string; messageId: string } + | { type: "clear" }; + +function transition( + state: BroadcastStreamState, + event: BroadcastStreamEvent +): TransitionResult; +``` + +The machine handles accumulator creation (including continuation context walking), chunk application, replay suppression, done/error cleanup, and produces `messagesUpdate` closures for the caller to apply. Side effects (sending ACKs, calling `onData`, `setIsServerStreaming`) stay in the caller. + +**Scope**: covers only the broadcast/resume accumulator path (path B). The transport-owned path (path A — local tab requests via `WebSocketChatTransport`) is managed by the AI SDK's `useChat` and doesn't go through the state machine. The transport's resume resolver state (`_resumeResolver`, `_resumeNoneResolver`, `_expectToolContinuation`) stays in `ws-chat-transport.ts`. + +**What still uses independent variables**: `localRequestIdsRef` (path A vs B switch), `resumingToolContinuationRef` (tool continuation re-entrancy guard), `useChatHelpers.status` (AI SDK lifecycle), and the transport's resolver state. These cross-cut the broadcast/transport boundary and aren't part of the accumulator lifecycle. + +## History + +- This design doc was created alongside the initial shared layer extraction. +- No prior RFCs — the extraction was motivated by Think's fork of `message-builder.ts` and the growing complexity of `ai-chat/src/index.ts`. +- TurnQueue extracted to `agents/chat/turn-queue.ts`. AIChatAgent and Think both adopt it, unifying turn serialization and the epoch/clear-generation concept. +- Broadcast stream state machine extracted to `agents/chat/broadcast-state.ts`. `useAgentChat`'s `onAgentMessage` handler uses `broadcastTransition` instead of manual accumulator/ref management. +- Think stripped to minimal core: single-session inline storage, removed multi-session API, deleted `AgentChatTransport`, disconnected extensions from Think class. Session module and transport deleted. +- ResumableStream moved from ai-chat to `agents/chat/resumable-stream.ts`. Resume protocol constants (`STREAM_RESUMING`, `STREAM_RESUME_ACK`, `STREAM_RESUME_REQUEST`, `STREAM_RESUME_NONE`) added to `CHAT_MESSAGE_TYPES`. Think wired with full resume support. +- Client tool primitives (`ClientToolSchema`, `createToolsFromClientSchemas`) moved to `agents/chat/client-tools.ts`. Tool protocol constants (`TOOL_RESULT`, `TOOL_APPROVAL`, `MESSAGE_UPDATED`) added. Think implements client-side tools with debounce-based auto-continuation. +- Think now has: MCP `waitForMcpConnections`, message push on connect, feature parity with AIChatAgent's core chat experience. diff --git a/design/loopback.md b/design/loopback.md new file mode 100644 index 0000000000..eca554489f --- /dev/null +++ b/design/loopback.md @@ -0,0 +1,168 @@ +# Loopback Pattern + +Cross-boundary RPC for sub-agents and dynamic isolates. + +## The problem + +Sub-agents (facets) run as colocated child Durable Objects with their own isolated SQLite. The parent calls them via typed RPC stubs. But several things cannot cross the RPC boundary: + +- **`AbortSignal`** — not serializable without the `AbortSignal serialization` compat flag. Passing one from parent to sub-agent throws `DataCloneError`. +- **Closures and live objects** — an `RpcTarget` can cross the boundary, but it ties the child's execution to a live reference held by the parent. If the parent hibernates, the reference dies. +- **Dynamic worker bindings** — dynamic isolates loaded via `env.LOADER` can only receive `Fetcher`/`ServiceStub` in their `env`, not `RpcStub`. You cannot hand them an RPC handle to a Durable Object. +- **Persistable references** — a stub to a dynamic worker entrypoint cannot be persisted, because the system doesn't know how to restart the dynamic worker without help from whatever loaded it. + +These constraints show up in practice whenever a sub-agent needs to call back to the parent (or siblings), or when a dynamically loaded worker needs to interact with the agent that spawned it. + +## The pattern + +A **loopback** is a `WorkerEntrypoint` subclass that carries serializable props identifying a target, and resolves the actual target at call time via `ctx.exports`. + +```typescript +type LoopbackProps = { + agentId: string; + resourceId: number; +}; + +export class MyLoopback extends WorkerEntrypoint { + constructor(ctx: ExecutionContext, env: Env) { + super(ctx, env); + + // Resolve the target DO from ctx.exports using the serializable props + let ns = ctx.exports.MyAgent; + let stub = ns.get(ns.idFromString(ctx.props.agentId)); + + // Get the actual RPC target + let session = stub.getResource(ctx.props.resourceId); + + // Return a Proxy so callers see the loopback as the real thing + return new Proxy(session, { + get(target, prop, receiver) { + return Reflect.get(target, prop, target); + }, + getPrototypeOf() { + return WorkerEntrypoint.prototype; + } + }); + } + + // Workaround: at least one method must be declared or the runtime + // validator won't register the class and the binding won't be created. + _dummy() {} +} +``` + +Created via: + +```typescript +let loopback = ctx.exports.MyLoopback({ props: { agentId, resourceId } }); +``` + +This produces a `Fetcher` — the one type that can go into a dynamic isolate's `env`, be persisted by a gatekeeper, or be passed anywhere a service binding is accepted. + +### Why it works + +1. **Props are plain data** — `agentId` and `resourceId` are strings/numbers, fully serializable and persistable. +2. **Resolution happens at call time** — the constructor runs when someone invokes a method on the loopback. `ctx.exports` gives access to all DO namespaces in the same worker, so the loopback can find its target without holding a live reference. +3. **The Proxy is transparent** — callers interact with the loopback as if it were the real target. The `getPrototypeOf` override makes it pass `instanceof` checks against `WorkerEntrypoint`. +4. **Survives hibernation** — since there's no live reference to preserve, only serializable props, the loopback can be stored and re-resolved after the parent wakes. + +## Variations + +The pattern has several shapes depending on the direction of the call: + +### Parent-to-child binding (env injection) + +Place loopbacks in a dynamic isolate's `env` so it can call back to parent-managed resources: + +```typescript +// In the parent agent, when loading a dynamic worker: +let env = { + MY_RESOURCE: ctx.exports.ResourceLoopback({ + props: { agentId: this.ctx.id.toString(), resourceId: 42 } + }) +}; +return { mainModule: "worker.js", modules, env }; +``` + +The dynamic worker sees `env.MY_RESOURCE` as a normal service binding and calls methods on it. Each call triggers the loopback constructor, which resolves the parent DO and delegates. + +### Child-to-parent callback (hook delivery) + +When a child resource needs to call a hook exported by a dynamic worker, but the child can't hold a direct stub (not persistable), a loopback in the reverse direction works: + +```typescript +// The parent stores a loopback Fetcher on the child instead of a direct stub +let hookLoopback = ctx.exports.HookLoopback({ + props: { agentId: this.ctx.id.toString(), hookName: "onUpdate" } +}); +await child.setHook(hookLoopback); +``` + +The child calls the hook via the loopback. The loopback resolves the parent, which loads the dynamic worker and finds the hook entrypoint. + +### Tail worker delivery (log forwarding) + +Attach a loopback as a tail worker to a dynamic isolate to forward `console.log` output and exceptions back to the parent: + +```typescript +return { + mainModule: "worker.js", + modules, + tails: [ + ctx.exports.TailLoopback({ + props: { agentId: this.ctx.id.toString(), contextId: chatId } + }) + ] +}; +``` + +The tail loopback receives `TraceItem` events and delivers them to the parent via `ctx.exports`. + +## Relevance to the Agents SDK + +### Current state: ToolBridge / RpcTarget + +In the assistant example, the parent passes an `RpcTarget` (ToolBridge) to the sub-agent on each `chatWithBridge()` call. This works but has limitations: + +- The bridge is per-call — it must be passed as an argument every time +- `AbortSignal` cannot be passed alongside it (DataCloneError) +- If the parent hibernates while the sub-agent is mid-call, the RpcTarget reference dies +- The sub-agent cannot initiate calls back to the parent unprompted + +### Future direction: loopback bindings for sub-agents + +The loopback pattern could replace per-call RpcTarget passing with persistent, self-resolving bindings: + +```typescript +// Parent configures the sub-agent with loopback bindings at creation time +const session = await this.subAgent(ChatSession, "session-1", { + bindings: { + SHARED_WORKSPACE: ctx.exports.WorkspaceLoopback({ + props: { agentId: this.ctx.id.toString() } + }), + ABORT: ctx.exports.AbortLoopback({ + props: { agentId: this.ctx.id.toString(), requestId } + }) + } +}); +``` + +The sub-agent accesses these as `env.SHARED_WORKSPACE` — no need to pass them per-call. The abort loopback could expose a `poll()` method the sub-agent checks periodically, sidestepping the AbortSignal serialization issue entirely. + +This is speculative and depends on facets supporting custom env injection, which they do not today. But it's the direction the pattern points toward. + +## Tradeoffs + +- **Indirection cost** — every call through a loopback resolves the target DO from scratch. For facets (colocated children) this is cheap. For cross-worker calls it involves a network hop. +- **No type safety at the boundary** — the Proxy returns `any`. The caller doesn't get TypeScript type checking on the methods. This could be improved with a typed wrapper. +- **Constructor-return Proxy is unusual** — returning a Proxy from a constructor is a valid but surprising JavaScript pattern. It may confuse contributors unfamiliar with the codebase. +- **Runtime workarounds** — several `getOwnPropertyDescriptor` and `getPrototypeOf` overrides exist to work around workerd bugs. These should be removable as the runtime matures. + +## Origin + +This pattern was developed in the [Gadgets Workshop](https://github.com/nicholasblaskey/minions) backend (`packages/workshop-backend/src/overseer.ts`) where it is used extensively: + +- `GatekeeperLoopback` — injects gatekeeper session bindings into dynamic gadget workers +- `GatekeeperHookLoopback` — allows gatekeepers to call hooks exported by dynamic gadget workers +- `GadgetTailLoopback` — forwards console logs from dynamic gadgets back to the overseer +- `CodeModeTailLoopback` — forwards execution traces from one-shot code runs back to the overseer diff --git a/design/readonly-connections.md b/design/readonly-connections.md new file mode 100644 index 0000000000..3a4accd1da --- /dev/null +++ b/design/readonly-connections.md @@ -0,0 +1,204 @@ +# Readonly Connections + +This document describes the design of the readonly connections feature: what it does, the key decisions made, alternatives we considered, and known limitations. + +## Problem + +Agents are collaborative — multiple WebSocket clients connect to the same agent instance and share state. But not every client should be allowed to modify that state. A dashboard viewer shouldn't be able to change settings. A spectator in a game shouldn't be able to move pieces. A free-tier user shouldn't be able to trigger expensive mutations. + +We need a way to mark certain connections as "readonly" and enforce that restriction at the framework level, not in userland. + +## Design goals + +1. **Declarative** — developers declare _which_ connections are readonly, not _how_ enforcement works +2. **Enforcement at the framework boundary** — readonly checks happen inside `setState()`, so they can't be bypassed by forgetting a check in a callable +3. **No boilerplate** — no manual permission checks needed in every `@callable()` method +4. **Survives hibernation** — readonly status persists when the Durable Object goes to sleep and wakes up +5. **Invisible to user code** — the internal flag can't be accidentally read, overwritten, or leaked through `connection.state` + +## API surface + +### Server-side (Agent class) + +| Method | Purpose | +| ---------------------------------------------- | ------------------------------------------------------- | +| `shouldConnectionBeReadonly(connection, ctx)` | Hook called on connect. Return `true` to mark readonly. | +| `setConnectionReadonly(connection, readonly?)` | Dynamically change readonly status at any time. | +| `isConnectionReadonly(connection)` | Check a connection's current readonly status. | + +### Client-side (useAgent / AgentClient) + +| Option | Purpose | +| --------------------------- | --------------------------------------------------- | +| `onStateUpdateError(error)` | Callback when client-side `setState()` is rejected. | + +RPC errors from blocked callables surface as rejected promises from `agent.call()`. + +## How enforcement works + +Readonly is enforced in two places: + +### 1. Client-side `setState()` — in the message handler + +When a client sends a `CF_AGENT_STATE` message, the `onMessage` wrapper checks `isConnectionReadonly(connection)` before processing it. If readonly, the server sends back a `CF_AGENT_STATE_ERROR` message and does **not** call `_setStateInternal`. + +This path handles: `agent.setState(newState)` from client code (React hook, PartySocket, etc.). + +### 2. Server-side `setState()` — in the public method + +When a `@callable()` method calls `this.setState()`, the public `setState()` method checks `agentContext.getStore()` for the current connection. If the connection is readonly, `setState()` throws `Error("Connection is readonly")`. + +The error propagates through the RPC handler's try/catch and is sent back as an RPC error response (`{ success: false, error: "Connection is readonly" }`). + +This path handles: any `@callable()` that calls `this.setState()` internally. + +### Why `setState()` and not the RPC handler? + +We considered four options for blocking mutations from readonly connections: + +| Approach | Pros | Cons | +| ---------------------------------------------------- | -------------------------------------------------------------------------------------- | -------------------------------------------------------- | +| **A. Manual checks in each callable** | Works today, explicit | Boilerplate, easy to forget, security hole if missed | +| **B. `@callable({ mutates: true })` decorator flag** | Declarative per-method | Opt-in — developers have to remember to tag methods | +| **C. `shouldAllowRPC(connection, method)` hook** | Maximum flexibility | More work for developers, whitelist vs blacklist footgun | +| **D. Check inside `setState()`** | Single enforcement point, no decorator changes, read-only callables work automatically | Side effects before `setState()` still run (see Caveats) | + +We chose **D** because it matches the mental model: "readonly" means "cannot change state." A readonly connection can still call RPCs that _read_ data — it just can't write anything. The framework enforces this automatically without requiring any annotation on callable methods. + +### Why `setState()` and not `_setStateInternal()`? + +There are two paths into state mutation: + +1. **Client-side** — arrives as a `CF_AGENT_STATE` message, already has its own readonly guard before calling `_setStateInternal(state, connection)` +2. **Server-side** — `this.setState(state)` calls `_setStateInternal(state, "server")` + +Putting the check in `setState()` keeps each entry point responsible for its own access control: + +- Client message handler → checks readonly → calls `_setStateInternal` +- `setState()` → checks readonly via context → calls `_setStateInternal` +- `_setStateInternal` → focuses on validation (`validateStateChange`), persistence, and broadcast + +This also means `validateStateChange` (data validity) and the readonly check (access control) live at different levels. Access control comes first, before we even look at the data. + +The `state` getter also calls `_setStateInternal` for initialization (persisting `initialState` on first access). These are framework-level operations that must bypass the readonly check, which is another reason the check belongs in the public `setState()`, not in `_setStateInternal()`. + +### What about workflows? + +`_workflow_updateState` calls `this.setState()`. But workflows don't have a connection in `agentContext` — the store's `connection` is `undefined`. So the readonly check passes harmlessly. + +## Storage: connection state wrapping + +### Evolution + +The readonly flag storage went through three designs: + +1. **SQL table** (original PR) — `CREATE TABLE cf_agents_readonly_connections`. Worked but added schema, queries, and cleanup logic for a single boolean. +2. **`connection.setState({ _readonly: true })`** (first refactor) — leveraged partyserver's built-in per-connection state, which survives hibernation. Much simpler. But had a fatal flaw: any call to `connection.setState({ ... })` without the callback form would overwrite `_readonly`. +3. **Namespaced connection attachment** (current) — wraps `connection.state` and `connection.setState()` on each connection to hide the `_cf_readonly` key from user code. + +### How the wrapping works + +When the Agent first encounters a connection (in `onConnect` or `onMessage`), `_ensureConnectionWrapped(connection)` is called. This method: + +1. **Detects** whether `state` is an accessor property (getter) or a data property via `Object.getOwnPropertyDescriptor` +2. **Captures** raw state access — for accessor properties, it binds the original getter directly; for data properties, it snapshots the current value into a closure variable to avoid a circular reference after the override +3. **Stores** the raw accessors in a `WeakMap` (the `_rawStateAccessors` map) +4. **Overrides** `connection.state` (getter) to strip `_cf_readonly` from the returned value +5. **Overrides** `connection.setState` to preserve `_cf_readonly` when user code sets new state + +The accessor vs. data property distinction matters because partyserver defines `state` as a getter (via `Object.defineProperties`), but we also need to handle non-partyserver connections or future implementations where `state` might be a plain data property. Without this, the fallback `() => connection.state` would call our overridden getter after the property is replaced, creating an infinite loop. + +After wrapping: + +- `connection.state` returns everything **except** `_cf_readonly` +- `connection.setState({ myData: "foo" })` stores `{ _cf_readonly: , myData: "foo" }` in the raw attachment +- `connection.setState((prev) => ({ ...prev, count: 1 }))` receives `prev` without `_cf_readonly`, but the flag is merged back in +- `setConnectionReadonly` / `isConnectionReadonly` use `_rawStateAccessors` to read/write the flag directly + +### Why this required a partyserver change + +Partyserver defines `state` and `setState` on connection objects via `Object.defineProperties` — and prior to our patch, both properties had `configurable: false` (the default). This prevented us from redefining them with `Object.defineProperty`. + +The fix was a two-line change in partyserver: add `configurable: true` to both the `state` and `setState` descriptors in `createLazyConnection`. The default behavior is unchanged — `configurable` only means the property _can_ be redefined, not that it behaves differently. + +### Why `_cf_readonly` and not `_readonly`? + +The `_cf_` prefix namespaces the key to avoid collisions. Without it, a user storing `{ _readonly: false }` in their connection state would accidentally disable the feature. The prefix makes accidental collision vanishingly unlikely. The key name is defined once as the module-level constant `CF_READONLY_KEY` so it stays consistent across `_ensureConnectionWrapped`, `setConnectionReadonly`, and `isConnectionReadonly`. + +### Why not a completely separate namespace (e.g. `{ _cf: { ... }, _user: { ... } }`)? + +We considered storing all user state under a `_user` sub-key so there could be zero collision. But this breaks when user state is `null` or a primitive (you'd need to wrap it in an object). It also means MCP transport code — which stores `_standaloneSse` and `requestIds` in connection state — would need to be rewritten to use the `_user` namespace. + +The single-key approach (`_cf_readonly` alongside user keys) is simpler, handles all state types, and doesn't require changes to existing code that uses `connection.state`. + +### What about `getConnections()`? + +Connections returned by `getConnections()` are the same JavaScript objects that were wrapped in `onConnect`/`onMessage` (partyserver's `createLazyConnection` checks `isWrapped(ws)` and returns the existing wrapper). So our `Object.defineProperty` overrides persist. + +After hibernation, the Durable Object creates new wrapper objects for rehydrated WebSockets. The first `onMessage` call re-wraps them via `_ensureConnectionWrapped`. + +## Caveats + +### Side effects in callables still run + +The readonly check happens inside `this.setState()`, not at the start of the callable. If a method does work before calling `setState()`, that work still executes: + +```typescript +@callable() +async processOrder(orderId: string) { + await sendEmail(orderId); // runs + await chargePayment(orderId); // runs + this.setState({ ... }); // throws — but damage is done +} +``` + +The recommended pattern is to put the state write first: + +```typescript +@callable() +async processOrder(orderId: string) { + this.setState({ ... }); // throws immediately for readonly + await sendEmail(orderId); // only runs if setState succeeded + await chargePayment(orderId); +} +``` + +This is an inherent tradeoff of enforcing at the `setState` level rather than at the RPC handler level. We chose this approach because it doesn't require developers to annotate every callable, and most callables are simple state machines where `setState` is the primary operation. For the rare case of callables with expensive side effects, the "state write first" pattern is straightforward. + +### Readonly is per-connection, not per-user + +There's no built-in mapping from readonly status to user identity. If a user opens two tabs — one readonly, one writable — they have full write access from the second tab. Authentication and authorization are the developer's responsibility; readonly connections are a transport-level primitive. + +### Readonly doesn't restrict `this.sql` or other side effects + +Only `this.setState()` is gated. A callable can still write to SQL, send emails, call external APIs, or do anything else. Readonly means "cannot change the agent's shared state" — it's not a general permission system. + +### HTTP requests bypass readonly entirely + +Readonly is a WebSocket concept. HTTP requests (`onRequest`, `agentFetch`, `getAgentByName` + `agent.fetch()`) run with `connection: undefined` in the agent context, so the `setState()` check always passes. + +This is by design: + +- **Callables are WebSocket-only** — there's no HTTP callable path. `routeAgentRequest` only handles WebSocket upgrades; plain HTTP falls through to `onRequest`. So clients can't invoke `@callable()` methods over HTTP. +- **`onRequest` is developer-authored** — unlike the WebSocket message handler (which has automatic setState/RPC processing), `onRequest` is entirely custom code. There's no framework behavior to gate. +- **HTTP requests are stateless** — there's no persistent "connection" to mark as readonly. Each request stands alone. Standard HTTP auth (tokens, headers, cookies) is the right tool here. + +If your `onRequest` handler calls `this.setState()`, it will always succeed. Protect HTTP endpoints with authentication/authorization in your `onRequest` implementation — this is standard practice and not something the readonly feature should absorb. + +A future extension could add `shouldRequestBeReadonly(request)` to set a flag in the agent context for HTTP requests too, but that's essentially HTTP middleware/auth, which most frameworks leave to the developer. + +## Testing + +Tests live in `packages/agents/src/tests/readonly-connections.test.ts` and cover: + +- `shouldConnectionBeReadonly` hook marking connections based on query params +- Client-side `setState()` blocked for readonly, allowed for writable +- Mutating RPCs (`incrementCount` → `this.setState()`) blocked for readonly +- Non-mutating RPCs (`getState`) allowed for readonly +- Mutating RPCs allowed for writable connections +- Dynamic readonly status changes at runtime +- State broadcasts reaching readonly connections (they can still observe) +- Readonly status restored after reconnection (hibernation survival) +- Multiple connections with mixed readonly states + +The test agent (`TestReadonlyAgent` in `agents/readonly.ts`) has `incrementCount` (mutating) and `getState` (non-mutating) callables, plus `checkReadonly` and `setReadonly` for dynamic status changes. diff --git a/design/retries.md b/design/retries.md new file mode 100644 index 0000000000..5a076a2658 --- /dev/null +++ b/design/retries.md @@ -0,0 +1,183 @@ +# Retries + +This document describes the retry system in the Agents SDK: how it works, where it is used, the key decisions made, and alternatives considered. + +## Problem + +Agents interact with external services and Cloudflare platform APIs that can fail transiently: Durable Object RPCs, Workflow operations, MCP server connections, and user-defined callbacks in queues and schedules. Without structured retries, every failure is either fatal or requires developers to hand-roll retry logic with inconsistent patterns. + +The `cloudflare/actors` library includes well-tested retry primitives (`tryN`, `jitterBackoff`, `isErrorRetryable`). We wanted to bring similar reliability to the Agents SDK while keeping the API surface small and the implementation internal until the patterns prove out. + +## Design goals + +1. **Internal first** — the retry primitives (`tryN`, `jitterBackoff`, `isErrorRetryable`, `validateRetryOptions`) live in `src/retries.ts` and are not re-exported from the package entry point. Only the `RetryOptions` type is re-exported for TypeScript consumers. The primitives are implementation details that can change without a breaking change. +2. **Public `this.retry()`** — a single user-facing method on the `Agent` class for ad-hoc retry logic. Thin wrapper over the internals. +3. **Per-call-site configurability** — `schedule()`, `scheduleEvery()`, and `queue()` accept an optional `{ retry?: RetryOptions }` parameter so developers can tune retry behavior per task. +4. **Backward compatible** — all new parameters are optional. Existing code works unchanged. Schema migrations use `ADD COLUMN IF NOT EXISTS` pattern. +5. **Sensible defaults** — 3 attempts, 100ms base delay, jittered exponential backoff. No configuration needed for the common case. +6. **Class-level defaults** — override defaults for an entire agent via `static options = { retry: { ... } }`, following the existing pattern for `hibernate`, `sendIdentityOnConnect`, etc. + +## Architecture + +### Core primitives (`src/retries.ts`) + +Three functions, one type: + +| Export | Purpose | +| ------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `RetryOptions` | Interface: `{ maxAttempts?, baseDelayMs?, maxDelayMs? }` | +| `jitterBackoff(attempt, baseDelayMs, maxDelayMs)` | Full Jitter backoff per [AWS blog post](https://aws.amazon.com/blogs/architecture/exponential-backoff-and-jitter/). Returns `random(0, min(2^attempt * base, max))`. | +| `tryN(n, fn, options?)` | Retry `fn` up to `n` total attempts. Accepts optional `shouldRetry` predicate to bail early on non-retryable errors. | +| `validateRetryOptions(options, defaults?)` | Eagerly validate retry config. When `defaults` are provided, resolves partial options against them before cross-field checks. | +| `isErrorRetryable(err)` | Returns `true` if the error has `retryable: true` but is not an overloaded Durable Object error. Follows [CF best practices](https://developers.cloudflare.com/durable-objects/best-practices/error-handling/). | + +`tryN` is the only retry loop. Everything else composes on top of it. + +### Input validation + +Validation is designed to fail fast and fail clearly: + +- **`validateRetryOptions(options, defaults?)`** runs at enqueue/schedule/`this.retry()` time. It checks individual field ranges, enforces integer `maxAttempts`, and validates cross-field constraints (`baseDelayMs <= maxDelayMs`) after resolving against defaults. This means `{ baseDelayMs: 5000 }` against default `maxDelayMs: 3000` throws immediately instead of failing minutes later at execution time. +- **`tryN` also validates** its inputs with `Number.isFinite()` checks — guarding against `NaN` and `Infinity` that could produce zero-delay retries. Error messages are consistent between both validation paths (e.g. "retry.maxAttempts must be >= 1"). + +### Public API + +**`this.retry(fn, options?)`** on the `Agent` class. Retries all errors by default. Accepts an optional `shouldRetry` predicate to bail early on non-retryable errors. The predicate signature is `(err: unknown, nextAttempt: number) => boolean` — the `nextAttempt` parameter enables attempt-aware retry decisions (e.g. "retry rate-limit errors up to 3 times but only once for everything else"). The predicate is defined as an intersection with `RetryOptions` rather than a separate type — this keeps the type surface minimal while making `shouldRetry` unavailable on `queue()`/`schedule()` via IDE autocomplete (since those accept plain `RetryOptions`). + +**`RetryOptions`** type is re-exported from the package for TypeScript consumers who want to type their options objects. + +**`{ retry?: RetryOptions }`** parameter on `queue()`, `schedule()`, and `scheduleEvery()`. Stored as JSON in a `retry_options TEXT` column. Read back at execution time and passed to `tryN`. Retry options are **validated eagerly** at enqueue/schedule time via `validateRetryOptions()` with class-level defaults as the second argument — invalid values like `maxAttempts: 0`, `baseDelayMs: -1`, or `baseDelayMs` exceeding the resolved `maxDelayMs` all throw immediately. + +### Integration points + +| Location | What is retried | `shouldRetry` | Defaults | +| --------------------------- | --------------------------- | ------------------ | ------------------------------- | +| `_flushQueue()` | Queue callback execution | All errors | 3 attempts, 100ms base, 3s max | +| Schedule alarm handler | Schedule callback execution | All errors | 3 attempts, 100ms base, 3s max | +| `terminateWorkflow()` | `instance.terminate()` | `isErrorRetryable` | 3 attempts, 200ms base, 3s max | +| `pauseWorkflow()` | `instance.pause()` | `isErrorRetryable` | 3 attempts, 200ms base, 3s max | +| `resumeWorkflow()` | `instance.resume()` | `isErrorRetryable` | 3 attempts, 200ms base, 3s max | +| `restartWorkflow()` | `instance.restart()` | `isErrorRetryable` | 3 attempts, 200ms base, 3s max | +| `sendEventToWorkflow()` | `instance.sendEvent()` | `isErrorRetryable` | 3 attempts, 200ms base, 3s max | +| MCP `_restoreServer()` | Server reconnection | All errors | Per-server config or 3/500ms/5s | +| MCP `establishConnection()` | Post-OAuth connection | All errors | Per-server config or 3/500ms/5s | + +Workflow operations use `isErrorRetryable` because they are DO RPC calls where we can distinguish transient errors from permanent failures. Queue/schedule callbacks and MCP connections retry all errors because the failure modes are broader and user-defined. + +MCP retry config is stored in the `server_options` JSON column alongside `client` and `transport` options, so it persists across hibernation. Developers configure it via `addMcpServer(name, url, { retry: { ... } })` or `registerServer(id, { ..., retry: { ... } })`. + +### Observability + +Queue and schedule retry attempts emit observability events: + +- `queue:retry` — emitted before each retry attempt in `_flushQueue()`, with `callback`, `id`, `attempt`, and `maxAttempts` in the payload. +- `schedule:retry` — emitted before each retry attempt in the schedule alarm handler, with the same payload shape. + +These events are only emitted for attempts > 1 (the first attempt is not a "retry"). They use the existing `this.observability?.emit()` pattern, so they appear in the observability stream alongside `schedule:execute` and other events. This enables users to monitor retry behavior in dashboards and logs. + +### Performance + +`_resolvedOptions` is cached after first access. Static options never change during the lifetime of a DO instance, so the resolved options object is computed once and reused. This avoids allocating a new object on every call to `_flushQueue`, schedule alarm handler, or `this.retry()`. + +Retry option parsing from SQLite rows uses a shared `parseRetryOptions()` helper and a `resolveRetryConfig()` helper to merge per-task options with class-level defaults. These are used by both `_flushQueue` and the schedule alarm handler, eliminating duplicated parsing logic. + +## Key decisions + +### Why full jitter, not equal jitter or decorrelated jitter? + +Full jitter (`random(0, cap)`) has the best p99 latency characteristics for high-contention scenarios according to the [AWS analysis](https://aws.amazon.com/blogs/architecture/exponential-backoff-and-jitter/). The implementation is also the simplest — a single `Math.random()` call. + +### Why `shouldRetry` instead of retrying everything? + +Durable Object overload errors (`retryable: true` + `overloaded: true`) should not be retried — they indicate the DO is rejecting work to protect itself. Retrying would make congestion worse. For workflow operations, we use `isErrorRetryable` to respect this. For user callbacks and MCP connections, we retry all errors because we cannot know what kind of error the user's code will throw. + +The internal `TryNOptions.shouldRetry` and the public `this.retry()`'s `shouldRetry` use the same name and compatible signatures: `(err: unknown, nextAttempt: number) => boolean`. The `nextAttempt` parameter allows attempt-aware decisions. Functions that only care about the error (like `isErrorRetryable`) work as-is — extra arguments are ignored. + +### Why store retry options in the DB instead of in memory? + +Schedules and queues survive agent restarts (hibernation). If retry options were in memory, they would be lost when the DO hibernates and wakes up. Storing them as JSON in a `retry_options TEXT` column ensures they persist alongside the task. + +### Class-level default retry config + +Retry defaults are part of the existing `static options` pattern on the Agent class: + +```typescript +class MyAgent extends Agent { + static options = { + retry: { maxAttempts: 5, baseDelayMs: 200, maxDelayMs: 5000 } + }; +} +``` + +This was added to `DEFAULT_AGENT_STATIC_OPTIONS` alongside `hibernate`, `sendIdentityOnConnect`, and `hungScheduleTimeoutSeconds`. The `_resolvedOptions` getter merges the user's partial overrides with built-in defaults and caches the result, so `{ retry: { maxAttempts: 10 } }` only overrides `maxAttempts` while keeping `baseDelayMs` and `maxDelayMs` at their defaults. + +The type system enforces this correctly: `AgentStaticOptions.retry` is typed as `RetryOptions` (all fields optional), while `ResolvedAgentOptions.retry` is `Required` (all fields required, filled from defaults). Per-call-site options always take priority over class-level defaults. + +### Why `this.retry()` retries all errors by default? + +The method is designed for user code — calling external APIs, fetching data, sending notifications. These operations fail with generic `Error` objects or network errors that do not have a `retryable` property. Requiring a `shouldRetry` predicate would add friction for the 90% case. + +For the 10% case where selective retry is needed, `this.retry()` accepts an optional `shouldRetry` predicate: + +```typescript +await this.retry( + async () => { + const res = await fetch(url); + if (!res.ok) throw new HttpError(res.status); + return res.json(); + }, + { + shouldRetry: (err, nextAttempt) => { + if (err instanceof HttpError && err.status >= 400 && err.status < 500) { + return false; // 4xx: don't retry + } + return true; // 5xx, network errors: retry + } + } +); +``` + +`shouldRetry` is only available on `this.retry()`, not on `schedule()`/`queue()`, because functions cannot be serialized to SQLite. For scheduled/queued tasks, handle non-retryable errors inside the callback. + +### Why not adopt `tryWhile` from cloudflare/actors? + +The actors library has a `tryWhile(condition, fn)` that retries as long as a condition function returns true. We dropped it because: + +1. `tryN` with `shouldRetry` covers the same use case more safely (bounded attempts) +2. `tryWhile` with a bug in the condition function retries forever +3. Internal-only usage does not need the flexibility + +## Schema migration + +Two columns added via `ALTER TABLE ... ADD COLUMN`: + +- `cf_agents_schedules.retry_options TEXT` — JSON-serialized `RetryOptions` +- `cf_agents_queues.retry_options TEXT` — JSON-serialized `RetryOptions` + +Both use the `addColumnIfNotExists` pattern already established for `intervalSeconds`, `running`, and `execution_started_at`. The migration runs in the Agent constructor alongside existing migrations. + +## Tradeoffs + +**Queue callbacks still dequeue on failure.** After all retry attempts are exhausted, the task is dequeued. We do not have a dead-letter queue. If a task fails permanently, it is logged and routed through `onError()`, but the item is removed. This matches the existing behavior (before retries, tasks were dequeued immediately after one attempt). A dead-letter mechanism could be added later. + +**No circuit breaker.** If an external service is down, retry attempts will consume wall-clock time (up to `maxAttempts * maxDelayMs`). For queue and schedule execution, this delays subsequent tasks. A circuit breaker pattern could short-circuit retries after repeated failures, but adds significant complexity. Deferred for now. + +**Retry delays block the event loop.** `tryN` uses `setTimeout` between attempts. During this time, the DO is awake but idle. For short delays (100ms–3s) this is acceptable. For longer delays, consider using `schedule()` to retry at a future time instead of blocking. Queue retries are head-of-line blocking — one failing item's retries delay all subsequent items. If independent retry is needed, use `this.retry()` inside the callback instead of per-task retry options. + +## Testing + +Unit tests in `packages/agents/src/tests/retries.test.ts`: + +- `jitterBackoff`: value range, increasing upper bound with attempt number +- `tryN`: success on first attempt, success after transient failures, exhaust attempts, `shouldRetry` bail-out, `shouldRetry` receives nextAttempt, attempt number passed to fn, invalid inputs (zero, NaN, Infinity), fractional n floors to integer, n=1 behavior +- `validateRetryOptions`: valid options, maxAttempts < 1, non-integer maxAttempts, non-finite maxAttempts, baseDelayMs/maxDelayMs <= 0, cross-field baseDelayMs > maxDelayMs, single-field without defaults, resolution against defaults +- `isErrorRetryable`: retryable non-overloaded errors, non-retryable errors, overloaded message variants, overloaded property, non-object errors + +Integration tests in `packages/agents/src/tests/retry-integration.test.ts`: + +- `this.retry()`: succeed on first attempt, succeed after transient failures, exhaust retries +- `shouldRetry`: transient errors succeed, permanent errors bail early, receives next attempt number +- `queue()` with retry: retries and succeeds, persists retry options on single items, persists retry options on multiple items via `getQueues` +- `schedule()` with retry: retries and succeeds, persists retry options +- Eager validation: rejects invalid options on queue/schedule, rejects cross-field invalid options resolved against defaults, rejects fractional maxAttempts +- Class-level defaults: uses class-level maxAttempts, exhausts after class-level maxAttempts, per-call override diff --git a/design/rfc-sub-agents.md b/design/rfc-sub-agents.md new file mode 100644 index 0000000000..670557dac4 --- /dev/null +++ b/design/rfc-sub-agents.md @@ -0,0 +1,205 @@ +# RFC: Sub-Agents + +Status: accepted + +## The problem + +A single Agent is one Durable Object with one SQLite database. That's fine for simple cases, but many real applications need internal structure: + +- **Isolation** — A code sandbox agent needs a database that the LLM cannot access directly. If the agent's own SQLite holds both the approval queue and the customer data, there's no structural enforcement — the LLM can bypass the queue by writing SQL. You need a separate storage boundary. + +- **Multiplicity** — A chat application needs many rooms, each with its own message history and LLM context. Stuffing all rooms into one SQLite with a `room_id` column works, but there's no isolation between rooms, no independent lifecycle, and the parent agent becomes a god object that manages every room's state. + +- **Parallel work** — An analysis agent wants to fan out a question to three specialist personas, each making independent LLM calls with their own system prompts and history. Running these sequentially is slow. Running them in parallel within a single agent means shared mutable state and no isolation between the personas. + +- **Bounded context** — A gatekeeper agent needs to enforce that all database mutations go through an approval queue. If the database lives in the same agent, enforcement is a convention ("don't call `this.sql` directly"). You want it to be structural — the agent literally has no path to the data except through a typed interface. + +All of these require the same primitive: child Durable Objects colocated with the parent, each with their own isolated SQLite, callable via typed RPC. The workerd runtime provides the building blocks (`ctx.facets`, `ctx.exports`), but the Agents SDK needs a first-class abstraction for this. + +## The design + +Sub-agent management is built directly into the `Agent` base class. There is no separate `SubAgent` class — any `Agent` can be mounted as either a top-level Durable Object (via wrangler bindings) or as a child facet (via `this.subAgent()`). The behavior adapts based on how the agent is instantiated. + +### API + +Three methods on `Agent`: + +```typescript +import { Agent } from "agents"; + +export class SearchAgent extends Agent { + onStart() { + this + .sql`CREATE TABLE IF NOT EXISTS cache (q TEXT PRIMARY KEY, result TEXT)`; + } + + async search(query: string): Promise { + const cached = this.sql`SELECT * FROM cache WHERE q = ${query}`; + if (cached.length) return cached; + // ... fetch, cache, return + } +} + +export class MyAgent extends Agent { + async doStuff() { + const searcher = await this.subAgent(SearchAgent, "main"); + const results = await searcher.search("hello"); + } +} +``` + +- **`subAgent(cls, name)`** — get or create a named child facet. Returns a typed RPC stub. The child class must extend `Agent` and be exported from the worker entry point. +- **`abortSubAgent(name, reason?)`** — forcefully stop a running child. Pending RPC calls receive the reason as an error. Transitively aborts the child's own children. The child restarts on the next `subAgent()` call. +- **`deleteSubAgent(name)`** — abort the child, then permanently wipe its storage. Transitively deletes the child's own children. Irreversible. + +Both parents and children use `Agent`. A child agent can itself call `this.subAgent()` to create nested facets. + +### `SubAgentStub` — typed RPC stubs + +When `this.subAgent(SearchAgent, "main")` returns, the result is a `SubAgentStub` — a mapped type that exposes all user-defined public methods as async RPC calls, while hiding `Agent` / `Server` / `DurableObject` internals. + +The exclusion uses `keyof Agent` — any method defined on `Agent` itself is hidden from the stub. This means new methods added to `Agent` are automatically excluded without maintaining a manual blocklist. Only user-defined methods on the subclass are exposed. + +### `SubAgentClass` — constructor type + +The `SubAgentClass` type uses `env: never` as a variance trick. Since `never` is assignable to every type, any `Agent` subclass satisfies the constraint regardless of its `Env` type parameter. The actual `env` is provided by the runtime when instantiating the facet, not by the caller. + +### Initialization + +`subAgent()` does two things: + +1. `ctx.facets.get(name, () => ({ class: exports[cls.name] }))` — creates or retrieves the facet +2. A set-name fetch (`/cdn-cgi/partyserver/set-name/`) — triggers `Server` initialization, which calls `onStart()` on first access + +The set-name fetch is the same pattern used by `getAgentByName` / `getServerByName`. It's a no-op if the child is already initialized. `onStart()` runs lazily on first `subAgent()` call, not eagerly on parent construction. + +### Validation + +The class name is checked against `ctx.exports` before attempting facet creation. If the class isn't exported from the worker entry point, a clear error is thrown: + +``` +Sub-agent class "Foo" not found in worker exports. +Make sure the class is exported from your worker entry point +and the export name matches the class name. +``` + +This catches the common mistake of forgetting to export the class, or using `export { Foo as Bar }` (which breaks the `cls.name` lookup). + +### Wiring + +Sub-agents do **not** need wrangler.jsonc entries — no bindings, no migrations. They are instantiated through `ctx.facets` and referenced via `ctx.exports`. The only requirement is that the class is exported from the worker entry point with its original name. + +## Patterns established + +Four `experimental/gadgets-*` examples demonstrate the API: + +### Fan-out / fan-in (`gadgets-subagents`) + +`CoordinatorAgent` (extends `AIChatAgent`) spawns three `PerspectiveAgent` sub-agents in parallel, each making independent LLM calls with different system prompts. Results are gathered via `Promise.all()` and synthesized. Each sub-agent persists its analysis history in its own SQLite. + +### Multi-room chat (`gadgets-chat`) + +`OverseerAgent` (extends `Agent`) manages a room registry. Each room is a `ChatRoom` sub-agent with its own message history and LLM context. The parent proxies WebSocket messages to the active room and manages stream relay between sub-agent and client. Deleting a room calls `this.deleteSubAgent()` — the sub-agent and its storage are permanently removed. + +### Isolated database (`gadgets-sandbox`) + +`SandboxAgent` (extends `AIChatAgent`) uses a `CustomerDatabase` sub-agent for data isolation. Dynamic Worker isolates (via Worker Loader) can only reach the database through a `DatabaseLoopback` WorkerEntrypoint that proxies back to the parent, which delegates to the sub-agent. Three layers of isolation: no network, single binding, sub-agent boundary. + +### Gated access (`gadgets-gatekeeper`) + +`GatekeeperAgent` (extends `AIChatAgent`) uses a `CustomerDatabase` sub-agent that the LLM cannot access directly. All mutations go through an approval queue. The sub-agent boundary makes this structurally enforceable — the agent has no path to the data except through the sub-agent's RPC methods. + +### The Loopback pattern + +When dynamic Worker isolates (from `env.LOADER`) need to call back to a sub-agent, they can't hold a sub-agent stub directly — they can only have `ServiceStub` bindings. The pattern is: + +1. Create a `WorkerEntrypoint` (e.g. `DatabaseLoopback`) that proxies to the parent Agent +2. The parent delegates to the sub-agent via `this.subAgent()` +3. Pass the WorkerEntrypoint as a binding to the dynamic isolate + +Chain: `dynamic isolate -> WorkerEntrypoint -> parent Agent -> sub-agent` + +## The alternatives considered + +### A. Separate `SubAgent` class + `withSubAgents` mixin (original proposal) + +The original design had a separate `SubAgent` base class for children and a `withSubAgents()` mixin to add management methods to parents. The rationale was to avoid requiring the `experimental` compat flag for users who don't use sub-agents. + +Rejected because: + +- **Two classes for the same thing** — `SubAgent` and `Agent` had nearly identical capabilities (both extended `Server`, both had `this.sql`, etc.). The distinction was confusing. +- **Mixin ergonomics were poor** — `const Parent = withSubAgents(AIChatAgent); export class MyAgent extends Parent` is awkward compared to just `extends AIChatAgent`. +- **The compat flag concern was overstated** — users who don't call `subAgent()` are unaffected by the methods existing on `Agent`. The `experimental` flag is only needed at runtime when `ctx.facets` is actually accessed. + +### B. Separate entry point without `experimental/` prefix + +Would suggest the API is stable. It isn't — it depends on `ctx.facets` and `ctx.exports`, which are behind the `experimental` compat flag in workerd. However, since the methods now live on `Agent` directly, the stability signal comes from the `@experimental` JSDoc tag on the methods rather than an import path. + +### C. Use `DurableObject` directly instead of extending `Server` + +Sub-agents could extend plain `DurableObject` instead of `Agent` (which extends `Server`). Lighter — no WebSocket machinery, no state sync, no MCP client. But: + +- `this.sql` is genuinely useful for sub-agents that store data (which is most of them) +- The set-name initialization pattern already exists in `Server` +- Since sub-agents are now just `Agent`, they get the full Agent feature set for free — scheduling, state sync, callable methods, etc. +- Consistency between parent and child reduces cognitive load +- The unused features have zero runtime cost until called + +### D. Allowlist instead of `keyof Agent` exclusion for `SubAgentStub` + +Instead of excluding `Agent` methods, we could require developers to register exposed methods. Rejected — the current approach (exclude everything on `Agent`, expose everything else) is zero-boilerplate and automatically adapts as `Agent` gains new methods. + +## Testing + +The sub-agent API has a full test suite in `packages/agents/src/tests/sub-agent.test.ts` covering: + +- Creation and RPC +- Persistence and isolation (parent and child have separate SQLite) +- Multiple sub-agents with independent state +- Abort and delete lifecycle +- Nested sub-agents (child spawning grandchild) +- Streaming callbacks via `RpcTarget` +- Missing export error guard +- Sub-agent name propagation + +Type-level tests in `packages/agents/src/tests-d/sub-agent-stub.test-d.ts` verify that `SubAgentStub` correctly exposes user methods and hides `Agent` internals. + +## Open questions + +### Graduating from `experimental` + +The methods are on `Agent` but marked `@experimental` in JSDoc. Graduation requires `ctx.facets` and `ctx.exports` leaving the `experimental` compat flag in workerd, plus sufficient real-world usage. + +### State sync between parent and sub-agent + +Sub-agents don't participate in the parent's `setState()` broadcast. If a sub-agent's data changes, the parent must explicitly re-sync. The gadgets examples handle this by calling `this.setState()` after sub-agent RPCs. A reactive pattern (sub-agent notifies parent of changes) might be worth exploring. + +### Cross-machine sub-agents + +Facets are colocated — the child runs on the same machine as the parent. A future extension could support remote sub-agents via standard DO stubs, but the API and failure modes would be very different. + +### Discovery and introspection + +A parent has no way to list its active sub-agents or query their health. There's no `listSubAgents()` or `getSubAgentStatus(name)`. The parent must track its own children in its own storage. + +### Resource limits + +There's no cap on how many sub-agents a parent can spawn, how deep the nesting can go, or how much total storage the tree consumes. Workerd may impose its own limits, but the SDK doesn't surface or enforce them. + +## Unsolved problems + +### Orchestration + +No framework-level support for coordinating sub-agents. The parent is responsible for fan-out/fan-in, error handling, and result synthesis. The gadgets examples hard-code these patterns. A general orchestration primitive doesn't exist yet. + +### Tracing and observability + +When a parent calls a sub-agent, which calls the LLM, which triggers a tool, which calls another sub-agent — there's no connected trace. Each sub-agent is an opaque RPC call. The `agents/observability` module has no awareness of the sub-agent tree. Needs trace ID propagation through facet calls. + +### Error propagation and resilience + +No retry logic, no circuit breaker, no structured error types for sub-agent failures. The retries design (`design/retries.md`) covers retry primitives but none are wired into sub-agent calls. + +## The decision + +Accepted. Sub-agent management methods (`subAgent`, `abortSubAgent`, `deleteSubAgent`) are built into the `Agent` base class. The separate `SubAgent` class and `withSubAgents` mixin have been removed. `SubAgentClass` and `SubAgentStub` types are exported from the main `agents` entry point. diff --git a/design/think-roadmap.md b/design/think-roadmap.md new file mode 100644 index 0000000000..1b3895ee0e --- /dev/null +++ b/design/think-roadmap.md @@ -0,0 +1,609 @@ +# Think Roadmap + +The implementation plan for `@cloudflare/think` — an opinionated chat agent base class built on Session for conversation storage and the Agents SDK for execution. + +This document synthesizes four prior analyses: + +- [think-sessions.md](./think-sessions.md) — Session integration design (the foundation) +- [think-vs-aichat.md](./think-vs-aichat.md) — feature gap analysis vs AIChatAgent (raw material — superseded by this doc for prioritization) +- [chat-api.md](./chat-api.md) — API analysis of AIChatAgent + useAgentChat (informs what to avoid) +- [chat-improvements.md](./chat-improvements.md) — non-breaking improvements + shared code extraction (feeds into Phase 1) + +Think hasn't shipped yet. There are no backward compatibility constraints. + +--- + +## Status + +| Phase | Description | Status | Commit | +| ----- | ---------------------------------------------------------- | -------- | ---------- | +| **0** | Shared extraction (`agents/chat`) + non-breaking additions | **Done** | `56558cd1` | +| **1** | Session integration into Think | **Done** | — | +| **2** | Regeneration (`regenerate-message` trigger) | **Done** | — | +| **3** | Programmatic API (`saveMessages`, `continueLastTurn`) | **Done** | — | +| **4** | Durability (`chatRecovery`, `onChatRecovery`) | **Done** | — | +| **5** | Polish (`messageConcurrency`, `resetTurnState`) | **Done** | — | + +**Phase 0 delivered:** `AbortRegistry`, `applyToolUpdate` + builders, `parseProtocolMessage` in `agents/chat`. `continuation` flag on `OnChatMessageOptions`. Tool part helpers, `getHttpUrl()`, `getAgentMessages()` in client layer. AIChatAgent refactored to use `AbortRegistry`. See [chat-improvements.md](./chat-improvements.md) for details. + +**Phase 1 delivered:** Session wired into Think as the storage layer. `this.messages` is now a getter backed by `session.getHistory()`. All storage internals removed (`_initStorage`, `_loadMessages`, `_appendMessage`, `_upsertMessage`, `_clearMessages`, `_deleteMessages`, `_rebuildPersistenceCache`, `_enforceMaxPersistedMessages`, `_persistedMessageCache`, `maxPersistedMessages`, `_storageReady`, `#configTableReady`, `_think_config` table, `think_request_context` table). Switched to `AbortRegistry`, `parseProtocolMessage`, `applyToolUpdate`/`toolResultUpdate`/`toolApprovalUpdate` from `agents/chat`. Added `configureSession()` override point, `onChatResponse()` lifecycle hook with re-entrancy guard, `ChatResponseResult` type, `continuation` flag on `ChatMessageOptions`, context tool auto-merge in `onChatMessage`, and `assembleContext()` returning `{ system, messages }` with context block composition. + +**Phase 2 delivered:** Non-destructive regeneration via `trigger: "regenerate-message"`. New responses branch from the same parent as the old response — old alternatives stay in the tree, accessible via `session.getBranches(parentId)`. `getHistory()` follows the latest leaf automatically. Contrast with AIChatAgent's destructive `_deleteStaleRows` approach. + +**Phase 3 delivered:** `saveMessages()` for programmatic turn entry (scheduled responses, webhooks, proactive agents) with function form and generation guards. `continueLastTurn()` for extending the last assistant response. Custom body persistence across hibernation, passed to `onChatMessage` in all turn paths (WebSocket, RPC, auto-continuation, programmatic). `sanitizeMessageForPersistence` hook for PII redaction and custom transforms. + +**Phase 5 delivered:** `messageConcurrency` strategies (queue/latest/merge/drop/debounce) matching AIChatAgent's feature set. Think's merge is non-destructive — all individual user messages stay in the Session tree, the model sees them all in one turn. `resetTurnState()` extracted as a protected method for subclasses. Drop check happens before `session.appendMessage` so dropped messages never touch the tree. + +**Phase 4 delivered:** `chatRecovery` flag wraps all 4 chat turn paths (WebSocket, auto-continuation, `saveMessages`, `continueLastTurn`) in `runFiber()` for durable execution. `_handleInternalFiberRecovery` override detects interrupted chat fibers. `onChatRecovery(ctx)` hook provides `ChatRecoveryContext` with partial text, stream chunks, recovery data (from `stash()`), and current messages. `_chatRecoveryContinue` scheduler waits for stable state then calls `continueLastTurn()`. `hasPendingInteraction()` and `waitUntilStable()` for quiescence detection. `_pendingInteractionPromise` for efficient wait-on-resolve. + +**All phases complete.** Think now has full feature parity with AIChatAgent plus Session-backed advantages (tree-structured messages, non-destructive regeneration, context blocks, compaction, FTS5 search). + +--- + +## Table of Contents + +1. [Architecture](#architecture) +2. [What Think Already Has](#what-think-already-has) +3. [What Think Has That AIChatAgent Doesn't](#what-think-has-that-aichatagent-doesnt) +4. [Remaining Gaps](#remaining-gaps) +5. [Deliberately Skipped](#deliberately-skipped) +6. [Implementation Plan](#implementation-plan) +7. [Client-Side Improvements](#client-side-improvements) +8. [Open Questions](#open-questions) + +--- + +## Architecture + +Think is three layers, plus a shared chat infrastructure layer consumed by both Think and AIChatAgent: + +``` +┌─────────────────────────────────────────────────────────┐ +│ Think │ +│ Chat execution: streaming, auto-continuation, │ +│ sub-agent RPC, extensions, configureSession │ +├─────────────────────────────────────────────────────────┤ +│ agents/chat │ +│ Shared primitives: TurnQueue, ResumableStream, │ +│ StreamAccumulator, AbortRegistry, tool state machine, │ +│ protocol handler, ContinuationState, sanitization │ +├─────────────────────────────────────────────────────────┤ +│ Session │ +│ Conversation data: tree messages, context blocks, │ +│ compaction, FTS5 search, multi-session, config │ +├─────────────────────────────────────────────────────────┤ +│ Agent │ +│ DO primitives: SQLite, WebSocket, RPC, scheduling, │ +│ fibers, MCP client, state sync │ +└─────────────────────────────────────────────────────────┘ +``` + +**`agents/chat`** is the shared chat infrastructure layer — consumed by both Think and AIChatAgent. It provides primitives for turn queue serialization, resumable streams, stream accumulation, abort management, tool state updates, protocol message handling, continuation state, and message sanitization. See [chat-improvements.md](./chat-improvements.md) for the extraction plan. + +**Session** owns all conversation data — messages, context blocks, compaction overlays, search indexes, configuration. See [think-sessions.md](./think-sessions.md) for the full design. + +**Think** owns the chat execution lifecycle — streaming to clients, auto-continuation, sub-agent RPC, extensions, and the `configureSession` builder. It imports shared primitives from `agents/chat` rather than reimplementing them. + +**Agent** provides the Durable Object primitives — SQLite, WebSocket hibernation, RPC, scheduling, fibers, MCP client. + +--- + +## What Think Already Has + +Features that are implemented or designed and ready to implement (via existing Think code + Session integration from [think-sessions.md](./think-sessions.md)): + +### Execution layer (Think-specific) + +| Feature | Source | Notes | +| --------------------- | ------------------------------ | ------------------------------------------------------------------------------ | +| Agentic loop | `onChatMessage` → `streamText` | Structured overrides: `getModel`, `getSystemPrompt`, `getTools`, `getMaxSteps` | +| Sub-agent RPC | `chat()` | `StreamCallback` interface for parent → child streaming | +| Dynamic configuration | `configure()` / `getConfig()` | Typed `Config` parameter, persisted in SQLite | +| Extensions | `ExtensionManager` | Sandboxed Worker tools, permission-gated, hot-loadable | +| Error handling | `onChatError` | Partial message persistence on failure | +| Auto-continuation | `_scheduleAutoContinuation` | 50ms coalesce, deferred queue | + +### Shared chat layer (`agents/chat` — used by both Think and AIChatAgent) + +| Feature | Source | Notes | +| -------------------- | ---------------------------------------- | --------------------------------------------------------------------------------------- | +| Turn queue | `TurnQueue` | Serial execution with generation-based invalidation | +| Resumable streams | `ResumableStream` | Chunk buffering in SQLite, replay on reconnect | +| Stream accumulation | `StreamAccumulator` | Build assistant message from stream chunks | +| Abort registry | `AbortRegistry` | Per-request `AbortController` management (extracted from both agents) | +| Tool state machine | `findAndUpdateToolPart` | Find tool part by `toolCallId`, match states, apply update (extracted from both agents) | +| Protocol handling | `ChatProtocolHandler` | WebSocket protocol message parsing and dispatch (extracted from both agents) | +| Continuation state | `ContinuationState` | Pending/active/deferred continuation tracking | +| Message sanitization | `sanitizeMessage`, `enforceRowSizeLimit` | Strip provider metadata, enforce 1.8MB row limit | +| Client tools | `createToolsFromClientSchemas` | Convert client-side tool schemas to AI SDK tools | +| Broadcast state | `broadcastTransition` | Client-side stream state machine for cross-tab broadcast | +| Request context | `RequestContextStore` | Key-value persistence for client tools, body, config (extracted from both agents) | + +See [chat-improvements.md §Shared Code Extraction](./chat-improvements.md#shared-code-extraction) for the extraction plan. Items marked "extracted from both agents" are currently duplicated between AIChatAgent and Think and will be consolidated into `agents/chat` before Think Phase 1. + +### Data layer (via Session integration) + +| Feature | Source | Notes | +| ------------------------- | ------------------------------- | ----------------------------------------------------------------- | +| Tree-structured messages | `AgentSessionProvider` | `parent_id` column, recursive CTE for history | +| `this.messages` getter | `session.getHistory()` | Always fresh, applies compaction overlays | +| Context blocks | `ContextBlocks` | Readonly, writable, skills (R2), search (FTS5) | +| Auto-generated tools | `session.tools()` | `set_context`, `load_context`, `search_context` | +| Compaction | `onCompaction` + `compactAfter` | Non-destructive overlays, hermes-style algorithm | +| FTS5 search | `AgentSessionProvider` | Per-message indexing, per-session and cross-session | +| System prompt composition | `assembleContext()` | Context blocks → frozen prompt, falls back to `getSystemPrompt()` | +| Read-time truncation | `truncateOlderMessages()` | Old tool outputs and long text truncated before LLM | +| Multi-session | `SessionManager` | Create, list, delete, rename, fork, usage tracking | +| Config storage | `assistant_config` table | Client tools, body, Think config — one table | + +### Override points + +| Method | Default | Purpose | +| --------------------------- | ----------------------------------- | ------------------------------------------------------ | +| `getModel()` | throws | Return the `LanguageModel` | +| `getSystemPrompt()` | `"You are a helpful assistant."` | Simple system prompt (fallback when no context blocks) | +| `getTools()` | `{}` | Server-side `ToolSet` | +| `getMaxSteps()` | `10` | Max tool-call rounds per turn | +| `configureSession(session)` | pass-through | Add context blocks, compaction, search | +| `assembleContext()` | context blocks + truncated history | Full control over what's sent to the LLM | +| `onChatMessage(options?)` | `streamText` with assembled context | Full control over inference | +| `onChatError(error)` | passthrough | Customize error handling | + +--- + +## What Think Has That AIChatAgent Doesn't + +These are genuine advantages — features Think provides that AIChatAgent does not: + +| Feature | Think | AIChatAgent | +| ---------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Context blocks** | `withContext("memory", ...)` with auto-wired SQLite providers. Model can read/write its own persistent memory via `set_context` tool. Supports readonly blocks, writable blocks, R2-backed skill collections (`load_context`), and FTS5 searchable blocks (`search_context`). | Not available. System prompt is static per-request. No LLM-writable persistent memory. | +| **Compaction** | `onCompaction(fn)` + `compactAfter(threshold)`. Non-destructive overlays — original messages preserved, LLM-generated summaries replace ranges at read time. Iterative updates on subsequent compactions. Token-budget tail protection with tool-group alignment. | Not available. `maxPersistedMessages` deletes old messages (lossy). | +| **Branching / regeneration** | Tree-structured messages via `parent_id`. Regeneration creates a sibling branch — both old and new responses preserved. `getBranches(messageId)` returns alternatives. `getHistory()` follows latest leaf. | Destructive: `_deleteStaleRows` removes old response, re-runs inference. No version history. | +| **FTS5 search** | Every message indexed on insert. Per-session `searchMessages(query)`. Cross-session `SessionManager.search(query)`. `session_search` tool for model self-search. | Not available. | +| **Multi-session** | `SessionManager` with create, list, delete, rename, fork, usage tracking. Namespaced context blocks per session. Cross-session search. | One conversation per DO instance. | +| **Structured overrides** | `getModel()`, `getSystemPrompt()`, `getTools()`, `getMaxSteps()`, `configureSession()`, `assembleContext()` — each has clear defaults and a single responsibility. Minimal subclass is 3 lines. | Single `onChatMessage(onFinish, options?)` override. Must wire `this.messages`, `convertToModelMessages`, `pruneMessages`, `streamText`, `toUIMessageStreamResponse` manually every time. | +| **Sub-agent RPC** | `chat(userMessage, callback, options?)` — parent agent drives sub-agent turns over Durable Object RPC with streaming via `StreamCallback`. | Not available. | +| **`onChatMessage` signature** | `(options?) → StreamableResult` — no unused `onFinish` callback, no HTTP `Response` abstraction mismatch. | `(onFinish, options?) → Response \| undefined` — `onFinish` is always a no-op internally, `Response` is consumed as a stream (never sent over HTTP). See [chat-api.md §S1](./chat-api.md#issue-s1-onchatmessage-signature-is-awkward). | +| **Skills from R2** | `R2SkillProvider` + `load_context` tool. Model sees skill metadata in system prompt, loads full content on demand. | Not available. | +| **Dynamic config** | `configure(config)` / `getConfig()` with typed `Config` parameter. Persisted across restarts. | Not available. | +| **Extension system** | `ExtensionManager` — sandboxed Worker tools loaded at runtime, permission-gated network/workspace access, hot-loadable via `load_extension` tool. | Not available. | +| **`assembleContext` returns `{ system, messages }`** | System prompt is composed from context blocks + compaction summaries, returned alongside model messages. Clean separation. | System prompt is inlined in the `streamText` call inside `onChatMessage`. No structured composition. | + +--- + +## Remaining Gaps + +Features from AIChatAgent that Think still needs. Session doesn't solve these — they're in the execution layer. + +### Gap 1: `continueLastTurn()` — Programmatic continuation + +**What it does:** Trigger a new LLM call that appends to the last assistant message rather than creating a new one. The LLM sees the full conversation (including the partial assistant response) and continues from where it left off. + +**Why Think needs it:** Building block for chat recovery (#2). Also enables "generate more" buttons, agent self-correction, and subclass-driven continuation. + +**Implementation with Session:** Session's tree structure makes this clean — the continuation appends as a child of the same parent the original assistant message was parented to. `getHistory()` follows the latest leaf, so the continued response replaces the interrupted one in the active path. Needs chunk rewriting (strip `messageId` from `start` chunks) so clients append to the existing message. + +**Depends on:** Nothing (can implement standalone). + +**Effort:** Medium. Needs turn queue integration, chunk rewriting, and a `continuation: true` flag wired through the pipeline. + +### Gap 2: `chatRecovery` / `onChatRecovery` — Durability + +**What it does:** Wraps every chat turn in `runFiber()`. If the DO is evicted mid-stream, the fiber recovers on restart: reconstructs partial response from stored chunks, calls `onChatRecovery()` for provider-specific recovery logic, then schedules `continueLastTurn()`. + +**Why Think needs it:** Long-running LLM calls (30–120+ seconds with tool chains) can be interrupted by DO eviction. Without fiber wrapping, the stream is lost with no recovery path. + +**Implementation with Session:** `onChatRecovery` receives a `ChatRecoveryContext` with `messages` from `session.getHistory()`, `partialText`/`partialParts` from stored chunks, and `recoveryData` from `this.stash()`. The `_chatRecoveryContinue` scheduler uses `waitUntilStable()` (#3) then calls `continueLastTurn()` (#1). `targetAssistantId` guard prevents stale continuations. + +**Depends on:** `continueLastTurn` (#1), `waitUntilStable` (#3). + +**Effort:** Medium-high. Four turn paths need fiber wrapping (WebSocket, auto-continuation, programmatic, `continueLastTurn`). Recovery pipeline needs `_handleInternalFiberRecovery` override, `ChatRecoveryContext` construction, and `schedule(0, "_chatRecoveryContinue")`. + +### Gap 3: `waitUntilStable()` / `hasPendingInteraction()` + +**What it does:** `hasPendingInteraction()` checks whether any message has a tool part in `input-available` or `approval-requested` state. `waitUntilStable()` combines turn queue drain with pending interaction polling, with a configurable timeout. + +**Why Think needs it:** Prerequisite for safe chat recovery — you can't continue if the client hasn't responded to a pending tool call. Also useful for programmatic agents and test harnesses. + +**Implementation with Session:** Check `session.getHistory()` for pending tool states. Session's `getHistory()` returns the current path including compaction overlays — tool states are always fresh. + +**Depends on:** Nothing. + +**Effort:** Low. Mostly logic, no storage changes. + +### Gap 4: `saveMessages()` — Programmatic turn entry + +**What it does:** Inject messages and trigger a model turn from within the agent — without a WebSocket request. Accepts static messages or a callback `(currentMessages) => newMessages`. Waits for active turns, persists, runs turn, returns `{ requestId, status }`. + +**Why Think needs it:** Proactive agents (scheduled responses), webhook-triggered turns, `onChatResponse` chaining, notification-driven interactions. The existing `chat()` method serves sub-agent RPC, but there's no equivalent for internal programmatic use. + +**Implementation with Session:** `session.appendMessage()` for persistence, then run a programmatic turn via the turn queue. Generation guards prevent stale turns after clear. + +**Depends on:** Nothing (can implement standalone, but most useful with `onChatResponse`). + +**Effort:** Medium. Needs turn queue integration and generation tracking. + +### Gap 5: `onChatResponse` — Post-turn lifecycle hook + +**What it does:** Called after every turn completion (WebSocket, `saveMessages`, auto-continuation) once the assistant message is persisted and the turn lock released. Receives `ChatResponseResult` with message, requestId, continuation flag, and status. Safe to call `saveMessages` from inside (re-entrancy guard prevents recursive hooks). + +**Why Think needs it:** Observability, analytics, chaining behavior, usage tracking via `SessionManager.addUsage()`, refreshing system prompt after context block changes. + +**Already designed in:** [think-sessions.md](./think-sessions.md) — API surface is defined. Implementation is straightforward. + +**Depends on:** Nothing. + +**Effort:** Low. Wire into existing turn completion paths. + +### Gap 6: Regeneration (`regenerate-message` trigger) + +**What it does:** Client sends `trigger: "regenerate-message"` with a truncated message list. Server deletes the old response and runs a fresh turn. + +**Why Think needs it:** Standard chat UI feature — users expect "regenerate" on responses. + +**Implementation with Session:** Session's branching makes this non-destructive. Instead of deleting the old response, append a new assistant message as a sibling branch (same `parentId`). `getHistory()` follows the latest leaf — the new response is the active path. Old response accessible via `getBranches()`. + +This is **better** than AIChatAgent's approach: alternatives are preserved, users can browse response versions, and there's no data loss. + +**Depends on:** Nothing. Session branching already works. + +**Effort:** Low-medium. Parse `trigger: "regenerate-message"` in `_handleChatRequest`, resolve parent ID from message list, run turn with branching semantics. + +### Gap 7: `continuation` flag on `ChatMessageOptions` + +**What it does:** Let `onChatMessage` know whether it's being called for a continuation (after tool results, after recovery, via `continueLastTurn`) vs a fresh user turn. + +**Why Think needs it:** Subclasses can adjust system prompts, select different models, skip expensive context assembly (RAG, memory retrieval), or log different metrics for continuations. + +**Status:** Resolved in Phase 0. The `continuation` field is added to AIChatAgent's `OnChatMessageOptions` as a non-breaking addition (see [chat-improvements.md §3](./chat-improvements.md#3-add-continuation-to-onchatmessageoptions)). Think imports the shared type from `agents/chat` and sets the flag at its call sites. + +**Effort:** Zero for Think — the type and AIChatAgent wiring land in Phase 0. + +### Gap 8: `sanitizeMessageForPersistence` hook + +**What it does:** User-overridable transform called after built-in sanitization but before Session persistence. For redacting PII, custom compaction, stripping internal metadata. + +**Depends on:** Nothing. + +**Effort:** Very low. Insert hook call in `_persistAssistantMessage` before `session.appendMessage`/`updateMessage`. + +### Gap 9: Custom body persistence + +**What it does:** Persist the client's custom `body` fields from the chat request. Available in auto-continuations, programmatic turns, and recovery. Survives hibernation. + +**Implementation with Session:** Store in `assistant_config` table (same as client tools). + +**Depends on:** Nothing. + +**Effort:** Very low. Parse `body` from request, store in `assistant_config`, restore on turn start. + +### Gap 10: `messageConcurrency` strategies + +**What it does:** queue (default), latest, merge, drop, debounce for overlapping user submits. + +**Why it might matter:** Real-time typing UIs, rapid-fire interactions, search-as-you-type patterns. + +**Depends on:** Nothing. + +**Effort:** Medium. Significant logic but self-contained. + +**Status:** Demand-driven. Don't implement until users request it. + +### Gap 11: `resetTurnState` (protected) + +**What it does:** Expose turn reset as a protected method for subclasses that need to intercept or customize clear behavior. + +**Depends on:** Nothing. + +**Effort:** Very low. Extract from `_handleClear`. + +--- + +## Deliberately Skipped + +Features from AIChatAgent that Think will **not** implement, with rationale: + +| Feature | Rationale | +| ---------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `onFinish` callback on `onChatMessage` | Think's signature is cleaner without it. Use `onChatResponse` for post-turn metadata. See [chat-api.md §S1](./chat-api.md#issue-s1-onchatmessage-signature-is-awkward). | +| `Response` return type | Think uses `StreamableResult` (`toUIMessageStream()`). No HTTP abstraction mismatch. See [chat-api.md §S4](./chat-api.md#issue-s4-response-return-type-couples-to-http-semantics). | +| v4 → v5 message migration | Think is v5-only. No legacy clients. | +| Client message sync (`CF_AGENT_CHAT_MESSAGES` from client) | Session's idempotent `appendMessage` + tree structure handles reconnect scenarios. Full array sync is unnecessary. | +| Plaintext response support | `StreamableResult` is the right abstraction. Subclasses that want to return plain text can wrap it in a simple helper. | +| `maxPersistedMessages` | Replaced by compaction. Compaction is non-destructive and preserves information. | +| `_persistedMessageCache` | Session's `appendMessage` is idempotent by ID. `updateMessage` is explicit. No skip-unchanged optimization needed. | +| Message reconciliation (`reconcileMessages`) | Session's tree structure with idempotent append and explicit `updateMessage` handles the cases reconciliation was designed for (ID conflicts, tool state merge). | + +--- + +## Implementation Plan + +### Phase 0: Shared Extraction (prerequisite) + +**Goal:** Extract duplicated code from AIChatAgent into `agents/chat` so Think can import shared primitives rather than reimplementing them. These PRs land on AIChatAgent first (non-breaking refactors), then Think consumes the extracted modules. + +See [chat-improvements.md §Shared Code Extraction](./chat-improvements.md#shared-code-extraction) for the full extraction plan with side-by-side code comparison. + +| Extraction | What moved to `agents/chat` | Impact on Think Phase 1 | +| ---------------------------- | ------------------------------------------------------------- | -------------------------------------------------------------------------- | +| `AbortRegistry` | `Map` + get/cancel/remove/destroyAll | Think imports instead of building `_abortControllers` | +| `applyToolUpdate` + builders | `toolResultUpdate` / `toolApprovalUpdate` / `applyToolUpdate` | Think imports instead of writing `_applyToolResult` / `_applyToolApproval` | +| `parseProtocolMessage` | Typed parser for `cf_agent_chat_*` WebSocket messages | Think imports for type-safe protocol dispatch | + +**Each extraction is a non-breaking refactor to AIChatAgent** — the public API doesn't change, only the internal implementation moves to `agents/chat`. Think then imports from `agents/chat` in Phase 1. + +**Also in Phase 0 (non-breaking additions to AIChatAgent, see [chat-improvements.md §Non-Breaking Additions](./chat-improvements.md#non-breaking-additions)):** + +| Addition | Notes | +| ---------------------------------------- | ---------------------------------------------------------- | +| `continuation` on `OnChatMessageOptions` | Think reuses the shared type from `agents/chat` | +| Export `getAgentMessages()` | Works with Think agents (same protocol) | +| Export tool part helpers | Works with Think agents (same `UIMessage` format) | +| `getHttpUrl()` on `useAgent` | Removes `@ts-expect-error`, any Think client hook benefits | + +### Phase 1: Session Integration + +**Goal:** Wire Session into Think as the storage layer. This is the foundation — every subsequent phase builds on it. + +**Prerequisite:** Phase 0 extractions must be landed so Think can import from `agents/chat`. + +**Changes (from [think-sessions.md](./think-sessions.md)):** + +1. **`onStart` rewrite:** + - Remove `_initStorage`, `_loadMessages`, `_rebuildPersistenceCache` + - Create `Session.create(this)`, pass to `configureSession()` + - Store result as `this.session` + +2. **`this.messages` becomes a getter:** + + ```typescript + get messages(): UIMessage[] { + return this.session.getHistory(); + } + ``` + +3. **User message persistence:** + - Replace `_appendMessage(msg)` with `session.appendMessage(msg)` + - Remove `_appendMessage`, `_upsertMessage`, `_loadMessages` methods + +4. **Assistant message persistence:** + - Replace `_persistAssistantMessage` internals with `session.getMessage(id)` → `session.updateMessage(msg)` or `session.appendMessage(msg)` + - Remove `_persistedMessageCache`, `_rebuildPersistenceCache`, `_enforceMaxPersistedMessages`, `maxPersistedMessages` + +5. **`assembleContext` returns `{ system, messages }`:** + - Compose system prompt from context blocks (if configured) or `getSystemPrompt()` (fallback) + - Apply `truncateOlderMessages()` before `convertToModelMessages` + - `onChatMessage` destructures result, passes `system` and `messages` to `streamText` + +6. **Tool auto-merge:** + - `onChatMessage` merges `getTools()` + `clientTools` + `session.tools()` + `options.tools` + - Context tools (`set_context`, `load_context`, `search_context`) auto-included when context blocks configured + +7. **Clear:** + - Replace `_clearMessages()` + `this.messages = []` + `_persistedMessageCache.clear()` with `session.clearMessages()` + +8. **Config and client tools:** + - Use Session's `assistant_config` table directly for client tools and Think config + - Remove `_think_config` table, `think_request_context` table, `_storageReady`, `#configTableReady` + +9. **Protocol handling:** + - Use `parseProtocolMessage` from `agents/chat` (extracted in Phase 0) for typed protocol dispatch + - Think provides agent-specific handling per event type + - Simplifies `_setupProtocolHandlers` / `_handleProtocol` with type-safe switch + +10. **Abort management:** + - Use `AbortRegistry` from `agents/chat` (extracted in Phase 0) + - Replace `_abortControllers` Map + manual get/cancel/remove/destroyAll + +11. **Tool result/approval:** + - Use `applyToolUpdate` + `toolResultUpdate` / `toolApprovalUpdate` from `agents/chat` (extracted in Phase 0) + - Replace `_applyToolResult` / `_applyToolApproval` (~70 lines) with calls to shared functions + Think-specific persist/broadcast + +12. **Stream resume:** + - Think keeps its own resume handling (notify/ACK/replay) — this was not extracted in Phase 0 + - Uses existing `ResumableStream` from `agents/chat` (already shared) + +13. **Broadcast:** + - Session's `_emitStatus` broadcasts `CF_AGENT_SESSION` with token estimates and compaction status + - `_broadcastMessages` continues to use `this.messages` (now a getter) + - Resume exclusions handled by `StreamResumeHandler.getExclusions()` + +**Removes:** `_initStorage`, `_loadMessages`, `_appendMessage`, `_upsertMessage`, `_clearMessages`, `_deleteMessages`, `_rebuildPersistenceCache`, `_enforceMaxPersistedMessages`, `_persistedMessageCache`, `maxPersistedMessages`, `_storageReady`, `#configTableReady`, `think_request_context` table, `_think_config` table, `_applyToolResult`, `_applyToolApproval` (replaced by shared functions). + +**Adds:** `session` field, `configureSession()` override, `{ system, messages }` return from `assembleContext()`. + +**Imports from `agents/chat` (via Phase 0):** `AbortRegistry`, `applyToolUpdate`, `toolResultUpdate`, `toolApprovalUpdate`, `parseProtocolMessage`. + +### Phase 2: Regeneration + +**Goal:** First user-visible feature from Session's tree structure. + +**Note:** `onChatResponse` and `continuation` flag were originally planned for Phase 2 but were delivered in Phase 1 — they fell naturally out of the Session integration work. + +1. **Regeneration:** + - Parse `trigger: "regenerate-message"` in the protocol handler + - Find the user message that the old response branches from + - Run `onChatMessage` — new assistant message appends as sibling branch + - `getHistory()` follows latest leaf automatically + - Broadcast updated messages + +### Phase 3: Programmatic API + +**Goal:** Enable agents to drive their own turns without WebSocket requests. + +1. **`saveMessages()`:** + - Accept `UIMessage[] | ((currentMessages) => UIMessage[])` + - Wait for idle via turn queue + - Persist via `session.appendMessage` + - Run programmatic turn (no `connection` in `agentContext`) + - Generation guards for stale-after-clear detection + - Return `{ requestId, status }` + +2. **`continueLastTurn()`:** + - Find last assistant message via `session.getLatestLeaf()` + - Run `onChatMessage` with `continuation: true` + - Continuation chunk rewriting: strip `messageId` from `start` chunks + - Return `{ requestId, status }` + +3. **Custom body persistence:** + - Parse `body` from chat request (everything except `messages`, `clientTools`, `trigger`) + - Persist to `assistant_config` as `_lastBody` + - Restore on turn start, pass to `onChatMessage` as `options.body` + - Available in auto-continuations and `continueLastTurn` + +4. **`sanitizeMessageForPersistence` hook:** + - Called in `_persistAssistantMessage` after built-in sanitization, before Session persistence + - Default: passthrough + +### Phase 4: Durability + +**Goal:** Streams survive DO eviction. + +1. **`chatRecovery` flag:** + - Boolean property (default `false`) + - When `true`, wrap all turn paths in `runFiber(CHAT_FIBER_NAME:requestId, ...)` + - Four paths: WebSocket turns, auto-continuation, `saveMessages`, `continueLastTurn` + - `stash()` available during streaming for provider-specific checkpoint data + +2. **`waitUntilStable()` / `hasPendingInteraction()`:** + - `hasPendingInteraction()`: check `session.getHistory()` for tool parts in `input-available` or `approval-requested` state + - `waitUntilStable({ timeout? })`: drain turn queue + poll pending interactions with deadline + +3. **`onChatRecovery(ctx)` / `_chatRecoveryContinue`:** + - `_handleInternalFiberRecovery` override: detect `CHAT_FIBER_NAME:` prefix + - Build `ChatRecoveryContext` from `session.getHistory()` + stored stream chunks + `stash()` snapshot + - Default `onChatRecovery()` returns `{}` → persist partial + schedule continuation + - `_chatRecoveryContinue`: `waitUntilStable(10s)` → `targetAssistantId` guard → `continueLastTurn()` + - Use `schedule(0, "_chatRecoveryContinue", { targetAssistantId }, { idempotent: true })` + +### Phase 5: Polish (demand-driven) + +Implement when users request: + +1. **`messageConcurrency` strategies** — queue/latest/merge/drop/debounce +2. **`resetTurnState`** — extract from `_handleClear`, make protected + +### Timeline + +``` +Phase 0: Shared Extraction ──┐ Non-breaking refactors to AIChatAgent + ├─ AbortRegistry │ + non-breaking additions + ├─ RequestContextStore │ Can proceed in parallel with + ├─ findAndUpdateToolPart │ client-side improvements + ├─ ChatProtocolHandler │ + ├─ StreamResumeHandler │ + ├─ continuation flag │ + ├─ getAgentMessages() │ + ├─ tool part helpers │ + └─ getHttpUrl() │ + │ +Client-side improvements ───┤ Parallel track (see chat-improvements.md) + ├─ fallbackMessages │ + ├─ onChatError callback │ + └─ (future: non-suspending hook) │ + │ +Phase 1: Session Integration ───┤ Foundation — imports from agents/chat + │ +Phase 2: Regeneration ───┤ First user-facing win (branching) + │ +Phase 3: Programmatic API ───┤ saveMessages, continueLastTurn + │ +Phase 4: Durability ───┤ Fiber-wrapped turns, recovery + │ +Phase 5: Polish ───┘ Demand-driven +``` + +--- + +## Client-Side Improvements + +These improve `useAgentChat` for **all** agents (Think and AIChatAgent) and can proceed in parallel with server-side phases. Detailed analysis in [chat-api.md](./chat-api.md). + +### High priority + +| Issue | Description | Reference | +| ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Non-suspending hook** | Export `getAgentMessages()` for framework loaders. Add non-suspending variant with `{ isPending, error }`. Support `fallbackMessages` for instant conversation switching. | [chat-api.md §C1](./chat-api.md#issue-c1-suspense-only-initial-message-fetch), [#1011](https://github.com/cloudflare/agents/issues/1011), [#1045](https://github.com/cloudflare/agents/issues/1045) | +| **Tool UI components** | Export `` renderer with slot-based customization. Export `getToolPartState()` utility. Eliminate the tool-state-machine boilerplate every app reimplements. | [chat-api.md §C3](./chat-api.md#issue-c3-tool-ui-is-entirely-user-rebuilt-every-time) | +| **Combined hook** | `useAgentChat({ agent: "ChatAgent", name: "session-1" })` that manages the WebSocket connection internally. Keep split hooks for advanced use. | [chat-api.md §C2](./chat-api.md#issue-c2-two-hook-calls-required-for-basic-setup) | + +### Medium priority + +| Issue | Description | Reference | +| ----------------------------- | ------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------- | +| **Unified streaming status** | Make `isStreaming` the primary API. Simplify `status` / `isServerStreaming` / `isStreaming` confusion. | [chat-api.md §C4](./chat-api.md#issue-c4-isserverstreaming-vs-status-is-confusing) | +| **Structured error handling** | Add `onChatError` callback to hook options. Distinguish server vs network errors. | [chat-api.md §C5](./chat-api.md#issue-c5-no-structured-error-handling) | +| **Message rendering helpers** | `` component or `useMessageText`, `useToolParts` hooks. | [chat-api.md §X4](./chat-api.md#issue-x4-message-rendering-has-no-helpers) | + +### Low priority + +| Issue | Description | Reference | +| ------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- | +| **`addToolOutput` naming** | Align with AI SDK's `addToolResult` or clearly document distinction. | [chat-api.md §C6](./chat-api.md#issue-c6-addtooloutput-vs-addtoolresult-naming-confusion) | +| **Remove deprecated options** | `tools`, `experimental_automaticToolResolution`, `toolsRequiringConfirmation`, `autoSendAfterAllConfirmationsResolved` — plan major version removal. | [chat-api.md §X1](./chat-api.md#issue-x1-deprecated-options-accumulating) | +| **Remove PartySocket coupling** | Add public `getHttpUrl()` method, remove `@ts-expect-error` internal access. | [chat-api.md §C8](./chat-api.md#issue-c8-ts-expect-error-coupling-to-partysocket-internals) | + +### Think-specific client opportunities + +| Feature | Description | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| **Session status UI** | Session broadcasts `CF_AGENT_SESSION` with `{ phase, tokenEstimate, tokenThreshold }`. Client hook could expose `{ phase, tokenUsage }` for compaction progress bars, token counters. | +| **Context block UI** | Expose context blocks to the client. Display "Memory" block contents, show token usage per block, allow manual editing. | +| **Branching UI** | Session's `getBranches(messageId)` returns response alternatives. Client could show "← v1 / v2 / v3 →" navigation for regenerated responses. | +| **Conversation list** | `SessionManager.list()` returns `SessionInfo[]` with metadata. A `useConversations()` hook could power conversation sidebars without custom code. Addresses [chat-api.md §X2](./chat-api.md#issue-x2-no-conversation--session-management). | + +--- + +## Open Questions + +### Should `configureSession` be async? + +Context block providers like `R2SkillProvider` might need async init. Currently the builder is sync with lazy resolution — providers are loaded on first access (inside `_ensureReady()`). This works because `getHistory()`, `tools()`, `freezeSystemPrompt()`, etc. all call `_ensureReady()` before accessing providers. + +**Resolved:** `configureSession` accepts both sync and async return types (`Session | Promise`). `onStart` is async and awaits it. The Agent base class supports async `onStart` — it wraps and awaits the user's implementation. Sync implementations work unchanged (a sync return is a valid `Promise`). Async implementations can read from KV, D1, or R2 before configuring context blocks. + +### Should compaction run synchronously or in background? + +Currently `appendMessage` triggers auto-compaction synchronously when the token threshold is exceeded. For a chat turn, this adds LLM latency. + +**Options:** + +- **Sync (current):** Compaction runs inline. Simpler, but adds latency. +- **Background:** `schedule(0, "_compact")` defers to after the turn. No latency, but token count may briefly exceed threshold. +- **Post-turn:** Run compaction in `onChatResponse` after the turn lock is released. + +**Leaning:** Post-turn compaction in `onChatResponse`. The turn is complete, messages are persisted, and the compaction runs without blocking the next user interaction. If it fails, non-fatal — the conversation continues uncompacted. + +### How does regeneration interact with the wire protocol? + +The client sends `trigger: "regenerate-message"` with a truncated message list. AIChatAgent uses `_deleteStaleRows` to remove the old response. Think-on-Session uses branching instead — no deletion. + +**Question:** Does the client need to know about branching? Options: + +- **Transparent:** Server broadcasts updated `messages` (from `getHistory()`, which follows latest leaf). Client sees the new response replace the old one. `useAgentChat` works unchanged. +- **Branch-aware:** Server sends branch metadata. Client can show "v1 / v2" UI. Requires client-side changes. + +**Leaning:** Start transparent (no client changes needed). Add branch awareness later as a Think-specific client feature. + +### Multi-session + wire protocol + +When `session` is swapped based on `options.body.sessionId`, the existing WebSocket connections are still bound to the DO instance and receiving broadcasts. How should session switching work? + +**Options:** + +- **Per-request session:** Each `onChatMessage` call can use a different session via `options.body.sessionId`. Broadcasts go to all connections. Client is responsible for filtering by session. +- **Connection-scoped session:** Associate each WebSocket connection with a session ID on connect. Broadcasts are scoped to connections in the same session. +- **Separate DOs:** Each conversation is a separate DO instance (current model). `SessionManager` is only used for within-DO multi-session (sub-agent orchestration, branching). + +**Leaning:** Separate DOs for user-facing conversations. `SessionManager` for internal orchestration (sub-agent logs, branching, forking). Don't try to multiplex user-facing sessions in a single DO. + +### Should Session be promoted from experimental? + +Session is currently in `agents/experimental/memory/session`. If Think depends on it, Think inherits the "experimental" designation. Options: + +- **Promote Session** to `agents/memory/session` (or `agents/session`) — stable API, changeset required for changes. +- **Keep experimental** — Think is already `@experimental`, so inheriting experimental Session is consistent. +- **Inline** — Copy Session into `@cloudflare/think` as an internal module. Avoids cross-package coupling but duplicates code. + +**Leaning:** Promote Session to stable alongside Think's release. They ship together as a coherent system. diff --git a/design/think-sessions.md b/design/think-sessions.md new file mode 100644 index 0000000000..4da33737c1 --- /dev/null +++ b/design/think-sessions.md @@ -0,0 +1,1009 @@ +# Think + Session: Replacing `.messages` with Session + +> **Status: IMPLEMENTED.** This design was implemented in Phase 1. The integration is complete — `this.messages` is a getter backed by `session.getHistory()`, all storage internals have been removed, and Session is the sole storage layer. See [think-roadmap.md](./think-roadmap.md) for delivery details. + +Design for integrating `agents/experimental/memory/session` into Think as the conversation storage layer. Think hasn't shipped yet, so there is no backward compatibility constraint — this is a clean redesign. + +This is Phase 1 of the Think implementation plan. See [think-roadmap.md](./think-roadmap.md) for the full phased plan. + +Related: + +- [think.md](./think.md) — Think design doc +- [think-roadmap.md](./think-roadmap.md) — implementation plan (all phases complete) +- [think-vs-aichat.md](./think-vs-aichat.md) — feature gap analysis vs AIChatAgent (resolved) +- [chat-api.md](./chat-api.md) — API analysis of AIChatAgent + useAgentChat + +--- + +## Table of Contents + +1. [Motivation](#motivation) +2. [What Session Provides](#what-session-provides) +3. [What Gets Removed from Think](#what-gets-removed-from-think) +4. [New Architecture](#new-architecture) +5. [API Design](#api-design) +6. [Internal Changes](#internal-changes) +7. [Context Assembly Pipeline](#context-assembly-pipeline) +8. [Regeneration via Branching](#regeneration-via-branching) +9. [Multi-Session Support](#multi-session-support) +10. [Usage Examples](#usage-examples) +11. [Design Decisions](#design-decisions) +12. [What Session Doesn't Cover](#what-session-doesnt-cover) + +--- + +## Motivation + +Think currently stores messages in a flat `assistant_messages` table with no tree structure, no branching, no sessions, no compaction, and no context blocks. The design doc acknowledges this: + +> **Single conversation per instance.** Think currently stores all messages in a single flat table with no session ID. There is no multi-session support. The `SessionManager` from `agents/experimental/memory/session` is designed to fill this gap but has not been integrated. +> +> **No message reconciliation.** Think uses `INSERT OR IGNORE` for incoming messages — it does not handle the client sending edited or truncated message lists. +> +> **Compaction** — No (only in experimental Session) +> **Context blocks** — No (only in experimental Session) + +Session already provides all of this — tree-structured messages, compaction overlays, context blocks with provider system, FTS5 search, and multi-session management. Rather than reimplementing these features in Think, we integrate Session as Think's storage layer. + +Since Think hasn't shipped, we can make this the only storage design. No migration path, no compat layer. + +--- + +## What Session Provides + +### 1. Tree-structured messages with branching + +Messages have `parent_id`. `getHistory(leafId)` walks the tree root-to-leaf via recursive CTE. `getBranches(messageId)` returns sibling responses. This directly enables regeneration — a gap from the AIChatAgent comparison. + +```sql +-- Session's assistant_messages table +CREATE TABLE assistant_messages ( + id TEXT PRIMARY KEY, + session_id TEXT NOT NULL DEFAULT '', + parent_id TEXT, + role TEXT NOT NULL, + content TEXT NOT NULL, -- JSON-serialized UIMessage + created_at DATETIME DEFAULT CURRENT_TIMESTAMP +) +``` + +For linear conversations (the default), `getHistory()` returns the same flat `UIMessage[]` that Think's current `_loadMessages()` does. The tree structure only activates when `parentId` is explicitly used. + +### 2. Compaction overlays + +Stored in `assistant_compactions`, overlays are summaries that replace a range of messages at read time. The original messages remain in SQLite — the overlay is non-destructive: + +```typescript +// applyCompactions: replaces messages[fromId..toId] with a synthetic summary message +result.push({ + id: `compaction_${comp.id}`, + role: "assistant", + parts: [{ type: "text", text: comp.summary }], + createdAt: new Date() +}); +``` + +The `createCompactFunction` helper implements a full hermes-style algorithm: + +1. Protect head messages (first N) +2. Protect tail by token budget (walk backward from end) +3. Align boundaries to avoid splitting tool call/result groups +4. Summarize middle section with LLM (structured format with iterative updates) +5. Sanitize orphaned tool pairs after compaction + +Auto-compaction triggers when estimated token count exceeds a configurable threshold. + +### 3. Context blocks + +Persistent key-value blocks injected into the system prompt. Provider-based with four types: + +| Provider type | `get` | `set` | Extra | Tool | +| ------------------------- | ----- | -------- | --------------- | ---------------- | +| `ContextProvider` | Yes | — | — | — (readonly) | +| `WritableContextProvider` | Yes | Yes | — | `set_context` | +| `SkillProvider` | Yes | Optional | `load(key)` | `load_context` | +| `SearchProvider` | Yes | Optional | `search(query)` | `search_context` | + +Context blocks render into a frozen system prompt with structured headers: + +``` +══════════════════════════════════════════════ +SOUL +══════════════════════════════════════════════ +You are a coding assistant who writes clean TypeScript. + +══════════════════════════════════════════════ +MEMORY (Important facts — use set_context to update) [42% — 462/1100 tokens] +══════════════════════════════════════════════ +- User prefers functional patterns +- Project uses Cloudflare Workers +``` + +The frozen prompt is cached for LLM prefix caching — writes to context blocks update the provider immediately but don't invalidate the snapshot until `refreshSystemPrompt()` is called. + +### 4. FTS5 full-text search + +Every message is indexed in an FTS5 virtual table on insert/update. `searchMessages(query)` performs full-text search within a session. `SessionManager.search(query)` searches across all sessions. The model can search its own history via `session_search` or `search_context` tools. + +### 5. Session management (via `SessionManager`) + +Multi-session lifecycle: create, list, delete, rename, fork. Each session gets namespaced providers. Cross-session search. Usage tracking (input/output tokens, estimated cost). Session metadata (`parent_session_id`, `model`, `source`, `end_reason`). + +```typescript +interface SessionInfo { + id: string; + name: string; + parent_session_id: string | null; + model: string | null; + source: string | null; + input_tokens: number; + output_tokens: number; + estimated_cost: number; + end_reason: string | null; + created_at: string; + updated_at: string; +} +``` + +### 6. Token estimation and truncation utilities + +`estimateMessageTokens()` — heuristic token counting (hybrid char/word, no tokenizer dependency, ~80KB savings vs tiktoken). `truncateOlderMessages()` — read-time truncation of old tool outputs and long text. + +--- + +## What Gets Removed from Think + +Since Think hasn't shipped, these are not removals from a public API — they're simplifications of the internal design. + +### Removed: `_initStorage()` / flat `assistant_messages` table + +Think's current flat table: + +```sql +CREATE TABLE assistant_messages ( + id TEXT PRIMARY KEY, + role TEXT NOT NULL, + content TEXT NOT NULL, + created_at DATETIME DEFAULT CURRENT_TIMESTAMP +) +``` + +Replaced by Session's `AgentSessionProvider` which creates a richer table (with `session_id`, `parent_id`, FTS5 index, compaction table, config table). + +### Removed: `_persistedMessageCache` + +Think uses `Map` to skip unchanged SQL writes during streaming. Session's `appendMessage` is idempotent by ID (checks for existing before insert), and `updateMessage` is explicit — called only when the message content has actually changed. The persistence cache optimization is no longer needed. + +### Removed: `_loadMessages()` / `_appendMessage()` / `_upsertMessage()` / `_clearMessages()` / `_deleteMessages()` + +All five storage methods are replaced by Session equivalents: + +| Think (removed) | Session (replacement) | +| ---------------------- | ----------------------------- | +| `_loadMessages()` | `session.getHistory()` | +| `_appendMessage(msg)` | `session.appendMessage(msg)` | +| `_upsertMessage(msg)` | `session.updateMessage(msg)` | +| `_clearMessages()` | `session.clearMessages()` | +| `_deleteMessages(ids)` | `session.deleteMessages(ids)` | + +### Removed: `_rebuildPersistenceCache()` + +No cache, no rebuild. + +### Removed: `_enforceMaxPersistedMessages()` / `maxPersistedMessages` + +Compaction is the mechanism for managing conversation length. It preserves information as summaries instead of deleting messages. `maxPersistedMessages` was a lossy stopgap. + +### Removed: `this.messages` as a mutable field + +The mutable `messages: UIMessage[] = []` field becomes a computed getter backed by `session.getHistory()`. + +### Removed: `think_request_context` table + +Client tool persistence and other request context moves to Session's `assistant_config` table (which has `(session_id, key)` as primary key and was explicitly reserved for Think integration). + +### Removed: `_think_config` table + +Think's dynamic configuration (`configure()` / `getConfig()`) moves to `assistant_config` with a reserved key prefix. The `Config` type parameter and `configure()` / `getConfig()` API are preserved — only the backing table changes. + +### Removed: `_storageReady` / `#configTableReady` / `#configCache` + +Session handles its own table initialization lazily. Config moves to Session's config table. + +--- + +## New Architecture + +``` + Browser + | + WebSocket (cf_agent_chat_* protocol) + | + ┌───────┴───────┐ + │ Think │ + │ (top-level) │ + └───────┬───────┘ + | + ┌────────────┼────────────┐ + | | | + Session Agentic Loop Tools + (messages + (streamText) + context + | + compaction + ┌───┴────┐ + search + | Tools | + config) | MCP | + | Client | + | Ext. | + └────────┘ +``` + +Think owns the **chat execution lifecycle** — streaming, abort, client tools, resumable streams, WebSocket protocol, auto-continuation. Session owns the **conversation data** — message persistence, context blocks, compaction, search, multi-session management. + +--- + +## API Design + +### Class definition + +```typescript +export class Think< + Env extends Cloudflare.Env = Cloudflare.Env, + Config = Record +> extends Agent { + // ── Session ──────────────────────────────────────────────── + + /** The conversation session — messages, context, compaction, search. */ + session!: Session; + + /** + * Conversation history. Computed from the active session. + * Equivalent to `this.session.getHistory()`. + */ + get messages(): UIMessage[] { + return this.session.getHistory(); + } + + // ── Override points ──────────────────────────────────────── + + /** Return the language model to use for inference. */ + getModel(): LanguageModel; + + /** Return the system prompt (simple string). Used as fallback when no context blocks are configured. */ + getSystemPrompt(): string; + + /** Return the server-side tools for the agentic loop. */ + getTools(): ToolSet; + + /** Return the maximum number of tool-call steps per turn. */ + getMaxSteps(): number; + + /** + * Configure the session. Called once during `onStart`. + * Override to add context blocks, compaction, search, skills. + * + * The base session is pre-created with `Session.create(this)`. + * Return it with builder methods chained. + */ + configureSession(session: Session): Session | Promise; + + /** + * Assemble context for the LLM from the current session state. + * + * Default implementation: + * 1. Freezes the system prompt from context blocks (falls back to getSystemPrompt()) + * 2. Gets history from session + * 3. Applies read-time truncation (old tool outputs, long text) + * 4. Converts to model messages with tool call pruning + * + * Returns { system, messages } so the caller has both. + */ + async assembleContext(): Promise<{ + system: string; + messages: ModelMessage[]; + }>; + + /** + * Handle a chat turn. Default runs the agentic loop with assembled context. + * Override for full control over inference. + */ + async onChatMessage(options?: ChatMessageOptions): Promise; + + /** + * Called after a chat turn completes and the assistant message has been persisted. + * Override for logging, chaining, side effects. + */ + onChatResponse(result: ChatResponseResult): void | Promise; + + /** Handle an error during a chat turn. Override to customize. */ + onChatError(error: unknown): unknown; + + // ── Dynamic configuration ────────────────────────────────── + + /** Persist a typed configuration object. Survives restarts. */ + configure(config: Config): void; + + /** Read persisted configuration, or null if never configured. */ + getConfig(): Config | null; + + // ── Sub-agent RPC ────────────────────────────────────────── + + /** Run a chat turn via RPC from a parent agent. */ + async chat( + userMessage: string | UIMessage, + callback: StreamCallback, + options?: ChatOptions + ): Promise; + + // ── Message access ───────────────────────────────────────── + + /** Get all messages (alias for this.messages). */ + getMessages(): UIMessage[]; + + /** Clear all messages from the active session. */ + clearMessages(): void; +} +``` + +### `configureSession` — the builder pattern + +`configureSession` receives a pre-created `Session.create(this)` and returns it with builder methods applied. This is the primary configuration point for anything beyond the simple case: + +```typescript +configureSession(session: Session): Session { + return session; // Default: no context blocks, no compaction +} +``` + +The Session builder methods available: + +| Method | Purpose | +| ------------------------------- | --------------------------------------------------------- | +| `.withContext(label, options?)` | Add a context block (auto-wires to SQLite if no provider) | +| `.withCachedPrompt(provider?)` | Cache frozen system prompt in SQLite (or custom provider) | +| `.onCompaction(fn)` | Register LLM compaction function | +| `.compactAfter(tokenThreshold)` | Auto-compact when tokens exceed threshold | +| `.forSession(sessionId)` | Isolate to a session ID (for multi-session) | + +### `assembleContext` — system prompt composition + +The default `assembleContext` composes context blocks into the system prompt and returns both the system prompt and model messages: + +```typescript +async assembleContext(): Promise<{ system: string; messages: ModelMessage[] }> { + // 1. Get system prompt — context blocks if configured, else getSystemPrompt() + const hasContextBlocks = this.session.getContextBlocks().length > 0; + const system = hasContextBlocks + ? await this.session.freezeSystemPrompt() + : this.getSystemPrompt(); + + // 2. Get conversation history from session (tree walk + compaction overlays) + const history = this.session.getHistory(); + + // 3. Read-time truncation of old tool outputs and long text + const truncated = truncateOlderMessages(history); + + // 4. Convert to model messages with tool call pruning + const messages = pruneMessages({ + messages: await convertToModelMessages(truncated), + toolCalls: "before-last-2-messages" + }); + + return { system, messages }; +} +``` + +The return type changes from `Promise` to `Promise<{ system: string; messages: ModelMessage[] }>` so that context blocks can influence the system prompt. The default `onChatMessage` destructures this: + +```typescript +async onChatMessage(options?: ChatMessageOptions): Promise { + const baseTools = this.getTools(); + const clientToolSet = createToolsFromClientSchemas(options?.clientTools); + const contextTools = await this.session.tools(); + const tools = { ...baseTools, ...clientToolSet, ...contextTools, ...options?.tools }; + + const { system, messages } = await this.assembleContext(); + + if (messages.length === 0) { + throw new Error("No messages to send to the model."); + } + + return streamText({ + model: this.getModel(), + system, + messages, + tools, + stopWhen: stepCountIs(this.getMaxSteps()), + abortSignal: options?.signal + }); +} +``` + +Note: Session context tools (`set_context`, `load_context`, `search_context`) are auto-merged into the tool set. If no context blocks are configured, `session.tools()` returns `{}` — no overhead. + +### `getSystemPrompt` as the simple-case escape hatch + +For agents that don't need context blocks, `getSystemPrompt()` still works exactly as before: + +```typescript +export class SimpleAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + getSystemPrompt() { + return "You are a helpful assistant."; + } +} +``` + +No `configureSession` override needed. `assembleContext` detects no context blocks and uses `getSystemPrompt()` as the system prompt. Zero new concepts for the simple case. + +### `onChatResponse` — new lifecycle hook + +Think currently has no post-turn hook. Adding `onChatResponse` (borrowed from AIChatAgent but with a cleaner signature) provides: + +```typescript +export type ChatResponseResult = { + message: UIMessage; + requestId: string; + continuation: boolean; + status: "completed" | "error" | "aborted"; + error?: string; +}; +``` + +This fires after every turn completion — WebSocket, RPC, and auto-continuation. Useful for: + +- Usage tracking via `SessionManager.addUsage()` +- Logging, analytics, observability +- Triggering follow-up actions +- Refreshing the system prompt after context block changes + +--- + +## Internal Changes + +### `onStart` + +Before: + +```typescript +onStart() { + this._initStorage(); + this._resumableStream = new ResumableStream(this.sql.bind(this)); + this.messages = this._loadMessages(); + this._rebuildPersistenceCache(); + this._restoreClientTools(); + this._setupProtocolHandlers(); +} +``` + +After: + +```typescript +onStart() { + const baseSession = Session.create(this); + this.session = this.configureSession(baseSession); + + this._resumableStream = new ResumableStream(this.sql.bind(this)); + this._restoreClientTools(); + this._setupProtocolHandlers(); +} +``` + +No `_initStorage`, no `_loadMessages`, no `_rebuildPersistenceCache`. Session handles its own table creation lazily on first access. + +### User message persistence + +Before: + +```typescript +for (const msg of incomingMessages) { + this._appendMessage(msg); // INSERT OR IGNORE +} +this.messages = this._loadMessages(); +``` + +After: + +```typescript +for (const msg of incomingMessages) { + await this.session.appendMessage(msg); // Idempotent by ID, auto-parents to latest leaf +} +// No reload needed — this.messages is a getter that calls getHistory() +``` + +### Assistant message persistence (after streaming) + +Before: + +```typescript +private _persistAssistantMessage(msg: UIMessage): void { + const sanitized = sanitizeMessage(msg); + const safe = enforceRowSizeLimit(sanitized); + const json = JSON.stringify(safe); + + if (this._persistedMessageCache.get(safe.id) !== json) { + this._upsertMessage(safe); + } + + if (this.maxPersistedMessages != null) { + this._enforceMaxPersistedMessages(); + } + + this.messages = this._loadMessages(); +} +``` + +After: + +```typescript +private _persistAssistantMessage(msg: UIMessage): void { + const sanitized = sanitizeMessage(msg); + const safe = enforceRowSizeLimit(sanitized); + + const existing = this.session.getMessage(safe.id); + if (existing) { + this.session.updateMessage(safe); + } else { + this.session.appendMessage(safe); + } + // No reload — this.messages getter reads from session + // No maxPersistedMessages — use compaction instead +} +``` + +### Clear + +Before: + +```typescript +private _handleClear() { + this._turnQueue.reset(); + // ... abort all, clear resume, clear continuation ... + this._clearMessages(); + this.messages = []; + this._persistedMessageCache.clear(); + this._broadcast({ type: MSG_CHAT_CLEAR }); +} +``` + +After: + +```typescript +private _handleClear() { + this._turnQueue.reset(); + // ... abort all, clear resume, clear continuation ... + this.session.clearMessages(); + this._broadcast({ type: MSG_CHAT_CLEAR }); +} +``` + +### Client tool persistence + +Before (using `think_request_context` table): + +```typescript +private _persistClientTools(): void { + if (this._lastClientTools) { + this.sql` + INSERT OR REPLACE INTO think_request_context (key, value) + VALUES ('lastClientTools', ${JSON.stringify(this._lastClientTools)}) + `; + } else { + this.sql`DELETE FROM think_request_context WHERE key = 'lastClientTools'`; + } +} +``` + +After (using Session's `assistant_config` table): + +```typescript +private _persistClientTools(): void { + const sessionId = this.session._sessionId ?? ""; + if (this._lastClientTools) { + this.sql` + INSERT OR REPLACE INTO assistant_config (session_id, key, value) + VALUES (${sessionId}, 'lastClientTools', ${JSON.stringify(this._lastClientTools)}) + `; + } else { + this.sql`DELETE FROM assistant_config WHERE session_id = ${sessionId} AND key = 'lastClientTools'`; + } +} +``` + +### Dynamic configuration + +Before (using `_think_config` table): + +```typescript +configure(config: Config): void { + this._ensureConfigTable(); + this.sql`INSERT OR REPLACE INTO _think_config (key, value) VALUES ('config', ${json})`; +} +``` + +After (using `assistant_config` table): + +```typescript +configure(config: Config): void { + const sessionId = this.session._sessionId ?? ""; + this.sql` + INSERT OR REPLACE INTO assistant_config (session_id, key, value) + VALUES (${sessionId}, '_think_config', ${JSON.stringify(config)}) + `; + this.#configCache = config; +} +``` + +All Think-internal state consolidates into Session's `assistant_config` table. One table instead of three (`think_request_context`, `_think_config`, and the flat `assistant_messages`). + +### Broadcasting messages + +Before: + +```typescript +private _broadcastMessages(exclude?: string[]): void { + this._broadcast({ + type: MSG_CHAT_MESSAGES, + messages: this.messages + }, exclude); +} +``` + +After (same, but `this.messages` is now a getter): + +```typescript +private _broadcastMessages(exclude?: string[]): void { + this._broadcast({ + type: MSG_CHAT_MESSAGES, + messages: this.messages // Calls session.getHistory() + }, exclude); +} +``` + +--- + +## Context Assembly Pipeline + +The full pipeline from user message to LLM call: + +``` +1. User message arrives (WebSocket or RPC) + │ +2. session.appendMessage(userMsg) + │ → INSERT into assistant_messages (idempotent by ID) + │ → parent_id = latest leaf (or explicit) + │ → Index in FTS5 + │ → Auto-compact if over tokenThreshold + │ +3. Broadcast messages to other clients + │ +4. Enter turn queue + │ +5. assembleContext() + │ ├─ Context blocks configured? + │ │ ├─ Yes → session.freezeSystemPrompt() + │ │ │ → Check prompt store for cached prompt + │ │ │ → If none: load all block providers, render, cache + │ │ │ → Return frozen prompt string + │ │ └─ No → getSystemPrompt() + │ │ + │ ├─ session.getHistory() + │ │ → Recursive CTE: walk tree from latest leaf to root + │ │ → Apply compaction overlays (replace ranges with summaries) + │ │ → Return UIMessage[] + │ │ + │ ├─ truncateOlderMessages(history) + │ │ → Truncate old tool outputs (>500 chars) + │ │ → Truncate old long text (>10K chars) + │ │ → Keep recent messages intact + │ │ + │ └─ pruneMessages(convertToModelMessages(truncated)) + │ → Strip tool calls from older messages + │ → Return ModelMessage[] + │ +6. Merge tools: getTools() + clientTools + session.tools() + MCP tools + │ +7. streamText({ model, system, messages, tools, ... }) + │ +8. Stream result → broadcast chunks → persist assistant message + │ +9. onChatResponse({ message, requestId, status }) +``` + +--- + +## Regeneration via Branching + +Session's tree structure enables regeneration as a first-class feature — closing a major gap vs AIChatAgent. + +### How it works + +1. Client sends `trigger: "regenerate-message"` with a truncated message list (up to the point where the user wants to regenerate) +2. Think finds the user message that the old assistant response was parented to +3. A new `onChatMessage` turn runs — the model sees the same history up to the branch point +4. The new assistant message is appended with the same parent as the old response, creating a sibling branch +5. `getHistory()` follows the latest leaf by default — the new response is the active path + +``` +User: "Explain monads" + └─ Assistant (v1): "A monad is a monoid..." ← old response (still in tree) + └─ Assistant (v2): "Think of a monad as..." ← new response (latest leaf, active path) +``` + +The old response remains accessible via `getBranches(userMessageId)`. The client could show a "← Previous version" UI. + +### Contrast with AIChatAgent + +AIChatAgent handles regeneration by deleting stale rows (`_deleteStaleRows: true` in `persistMessages`). This is destructive — the old response is gone. Session's branching is non-destructive — all alternatives are preserved. + +--- + +## Multi-Session Support + +For agents that need multiple conversations per DO instance, Think can expose `SessionManager`: + +```typescript +export class MultiChatAgent extends Think { + sessions = SessionManager.create(this) + .withContext("memory", { description: "Learned facts", maxTokens: 2000 }) + .withSearchableHistory("history") + .withCachedPrompt(); + + configureSession(session: Session) { + // Default session — used when no sessionId is specified + return session + .withContext("memory", { maxTokens: 2000 }) + .withCachedPrompt(); + } + + // Switch session based on client request + async onChatMessage(options?: ChatMessageOptions) { + const sessionId = options?.body?.sessionId; + if (sessionId) { + this.session = this.sessions.getSession(sessionId); + } + return super.onChatMessage(options); + } +} +``` + +`SessionManager` provides: + +- `create(name)` — create a new session with metadata +- `list()` — list all sessions (ordered by `updated_at`) +- `delete(id)` — delete a session and its messages +- `rename(id, name)` — rename a session +- `fork(id, atMessageId, newName)` — fork a session at a specific point +- `search(query)` — cross-session FTS5 search +- `addUsage(id, inputTokens, outputTokens, cost)` — token accounting +- `compactAndSplit(id, summary)` — compact + archive old session, create new one with summary +- `tools()` — `session_search` tool for cross-session search + +This is an advanced feature. Most agents use the default single-session mode and never touch `SessionManager`. + +--- + +## Usage Examples + +### Minimal (unchanged from current Think) + +```typescript +export class ChatAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } +} +``` + +No `configureSession` needed. No context blocks. `getSystemPrompt()` returns the default. Behaves identically to a flat-message Think — Session operates in linear mode with no extras. + +### With system prompt (unchanged) + +```typescript +export class ChatAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + getSystemPrompt() { + return "You are a helpful coding assistant specializing in TypeScript."; + } +} +``` + +### With context blocks and self-updating memory + +```typescript +export class MemoryAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + + configureSession(session: Session) { + return session + .withContext("soul", { + provider: { get: async () => "You are a helpful coding assistant." } + }) + .withContext("memory", { + description: + "Important facts learned during conversation. Update proactively.", + maxTokens: 2000 + }) + .withCachedPrompt(); + } +} +``` + +The model gets: + +- A frozen system prompt with `SOUL` (readonly) and `MEMORY` (writable) blocks +- A `set_context` tool that writes to the `MEMORY` block +- The memory block persists across sessions and survives hibernation + +### With compaction for long conversations + +```typescript +import { createCompactFunction } from "agents/experimental/memory/utils"; + +export class LongChatAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + + configureSession(session: Session) { + return session + .withContext("memory", { maxTokens: 1500 }) + .onCompaction( + createCompactFunction({ + summarize: (prompt) => + generateText({ model: this.getModel(), prompt }).then((r) => r.text) + }) + ) + .compactAfter(50000) + .withCachedPrompt(); + } +} +``` + +When token count exceeds 50K, the session auto-compacts: summarizes the middle section with the LLM, stores an overlay, and refreshes the system prompt. The next `getHistory()` returns the compacted version. + +### With R2 skills (on-demand knowledge) + +```typescript +import { R2SkillProvider } from "agents/experimental/memory/session"; + +export class KnowledgeAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + + configureSession(session: Session) { + return session + .withContext("docs", { + description: "Project documentation — use load_context to read", + provider: new R2SkillProvider(this.env.DOCS_BUCKET, { prefix: "docs/" }) + }) + .withContext("memory", { maxTokens: 1500 }) + .withCachedPrompt(); + } +} +``` + +System prompt shows skill metadata (list of available docs). Model uses `load_context` to fetch specific docs on demand. + +### With usage tracking and post-turn hooks + +```typescript +export class TrackedAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + + configureSession(session: Session) { + return session + .withContext("memory", { maxTokens: 1500 }) + .withCachedPrompt(); + } + + async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed") { + // Refresh context blocks after each turn (memory may have been updated) + await this.session.refreshSystemPrompt(); + } + } +} +``` + +--- + +## Design Decisions + +### Why `configureSession` instead of constructor options? + +Builder methods are discoverable via IDE autocomplete. The method can reference `this.env` (runtime bindings aren't available at class definition time). Extension classes can call `super.configureSession(session).withContext(...)` to add blocks without replacing the parent's configuration. + +### Why `this.messages` as a getter instead of a cached field? + +A getter that calls `session.getHistory()` eliminates staleness bugs — messages are always fresh from SQLite. The cost is a recursive CTE per access, but for typical conversation lengths (10–200 messages) this is <1ms on DO SQLite. For the streaming hot path (where Think previously called `this.messages = this._loadMessages()` at turn start), the getter is called once per turn and the result used throughout — no different from the reload pattern. + +If profiling shows this is a bottleneck for very long conversations, we can add per-turn caching: set `_turnHistory = getHistory()` at turn start, return `_turnHistory` from the getter during a turn, clear at turn end. But we should profile first. + +### Why context blocks instead of extending `getSystemPrompt`? + +Context blocks solve three problems `getSystemPrompt` doesn't: + +1. **Persistence**: Blocks survive hibernation, restart, and eviction. `getSystemPrompt` is re-evaluated each time. +2. **LLM-writable**: The model can update its own context via `set_context`. With `getSystemPrompt` the model has no way to persist learned information. +3. **Prefix caching**: The frozen prompt is stable across turns (writes don't invalidate until `refreshSystemPrompt`), enabling LLM prefix cache hits. A dynamic `getSystemPrompt` that reads from storage on every call defeats prefix caching. + +### Why drop `maxPersistedMessages`? + +Compaction is strictly better — it preserves information as summaries instead of deleting messages. `maxPersistedMessages` was a blunt instrument: it deleted the oldest messages regardless of whether they contained important context. With compaction, the LLM generates a structured summary that preserves key information, decisions, and open items. + +For agents that truly need a hard ceiling (e.g., demo apps, resource-constrained environments), the compaction threshold serves the same purpose with better information retention. + +### Why not have Session own the streaming pipeline? + +Session is a storage and context layer. Streaming involves WebSocket protocol, chunk buffering, resumable streams, abort controllers, and broadcast — all of which are tightly coupled to Think's execution model. Keeping Session as pure storage makes it reusable (AIChatAgent could adopt it too) and keeps Think's streaming pipeline independent. + +### Why tree structure for all messages, not just regeneration? + +The `parent_id` column and recursive CTE add no overhead for linear conversations — `getHistory()` returns the same result as a flat `ORDER BY created_at`. But the tree structure enables: + +- **Regeneration**: branch at any point, keep alternatives +- **Forking**: `SessionManager.fork()` copies a conversation up to a point +- **Sub-agent responses**: a parent agent can branch the conversation to try different approaches +- **Undo**: remove the last branch, previous version becomes active + +These are all free once the tree is in place. Adding tree structure later would require a migration; starting with it costs nothing. + +--- + +## What Session Doesn't Cover + +These are handled by Think or the shared `agents/chat` layer. See [chat-improvements.md](./chat-improvements.md) for the extraction plan that moves duplicated code from AIChatAgent into `agents/chat` so Think can import rather than reimplement. + +| Concern | Owner | Notes | +| ----------------------- | -------------------------------------------------------- | ------------------------------------------------------------------ | +| Resumable streams | `agents/chat` (`ResumableStream`) | Chunk buffering in `cf_ai_chat_stream_*` tables | +| Stream resume handshake | Think | Notify/ACK/replay pattern (Think-specific, uses `ResumableStream`) | +| WebSocket protocol | `agents/chat` (`parseProtocolMessage`) | Typed parser for protocol dispatch | +| Abort/cancel | `agents/chat` (`AbortRegistry`) | Per-request `AbortController` management | +| Tool state updates | `agents/chat` (`applyToolUpdate` + builders) | State matching + update construction | +| Request context | Session (`assistant_config` table) | Client tools, body, config — Think uses Session directly | +| Auto-continuation | Think (`_continuation`, `_scheduleAutoContinuation`) | 50ms coalesce, deferred queue (Think-specific) | +| Stream accumulation | `agents/chat` (`StreamAccumulator`) | Build assistant message from chunks | +| Message sanitization | `agents/chat` (`sanitizeMessage`, `enforceRowSizeLimit`) | Applied before Session persistence | +| Turn queue | `agents/chat` (`TurnQueue`) | Serial turn execution with generation tracking | +| Continuation state | `agents/chat` (`ContinuationState`) | Pending/active/deferred tracking | +| Extensions | Think (`ExtensionManager`) | Sandboxed Worker tools (Think-specific) | +| Sub-agent RPC | Think (`chat()`) | `StreamCallback` interface (Think-specific) | + +### SQLite table ownership + +| Table | Owner | Purpose | +| ---------------------------- | -------------------------------- | ---------------------------------------------------- | +| `assistant_messages` | Session (`AgentSessionProvider`) | Tree-structured messages | +| `assistant_compactions` | Session (`AgentSessionProvider`) | Compaction overlays | +| `assistant_fts` | Session (`AgentSessionProvider`) | FTS5 full-text search index | +| `assistant_config` | Session (`AgentSessionProvider`) | Key-value config (Think config, client tools, etc.) | +| `assistant_sessions` | Session (`SessionManager`) | Multi-session metadata (only if SessionManager used) | +| `cf_agents_context_blocks` | Session (`AgentContextProvider`) | Context block storage | +| `cf_ai_chat_stream_metadata` | Think (`ResumableStream`) | Stream replay metadata | +| `cf_ai_chat_stream_chunks` | Think (`ResumableStream`) | Stream replay chunks | +| `cf_agents_runs` | Agent (inherited) | Durable fiber state | +| `cf_agents_schedules` | Agent (inherited) | Scheduled tasks | diff --git a/design/think-vs-aichat.md b/design/think-vs-aichat.md new file mode 100644 index 0000000000..986c8a0233 --- /dev/null +++ b/design/think-vs-aichat.md @@ -0,0 +1,275 @@ +# Think vs AIChatAgent + +A comparison of `@cloudflare/think` (`Think`) and `@cloudflare/ai-chat` (`AIChatAgent`) — two chat agent base classes built on the Agents SDK. Both extend `Agent` and speak the same `cf_agent_chat_*` WebSocket protocol, but they serve different goals. + +Related: + +- [think-roadmap.md](./think-roadmap.md) — Think implementation plan (all phases complete) +- [think-sessions.md](./think-sessions.md) — Session integration design +- [chat-api.md](./chat-api.md) — AIChatAgent + useAgentChat API analysis +- [chat-improvements.md](./chat-improvements.md) — shared extraction + client DX improvements + +--- + +## Philosophical difference + +**AIChatAgent is a protocol adapter.** It bridges the `cf_agent_chat_*` WebSocket protocol to the AI SDK. You override `onChatMessage(onFinish, options) → Response | undefined` — you're responsible for calling `streamText`, wiring up tools, converting messages, constructing the system prompt, and returning a `Response`. AIChatAgent handles the plumbing: message persistence, streaming, abort, resume, client sync. But the LLM call is entirely your problem. + +**Think is an opinionated framework.** It makes decisions for you: `getModel()` returns the model, `getSystemPrompt()` or `configureSession()` sets the prompt, `getTools()` returns tools, `assembleContext()` handles message conversion + truncation + pruning. The default `onChatMessage` runs the complete agentic loop. You override individual pieces, not the whole pipeline. + +--- + +## API surface comparison + +### Override points + +| Concept | AIChatAgent | Think | +| ------------------------- | --------------------------------------------------------------------------- | --------------------------------------------------------------- | +| **Minimal subclass** | ~15 lines (wire `streamText` + tools + messages + system prompt + response) | 3 lines (`getModel()` only) | +| **onChatMessage** | `(onFinish, options) → Response \| undefined` | `(options?) → StreamableResult` | +| **System prompt** | Inline in your `onChatMessage` | `getSystemPrompt()` or `configureSession()` with context blocks | +| **Tools** | Inline in your `onChatMessage` | `getTools()` + auto-merge with client tools + context tools | +| **Context assembly** | Manual in `onChatMessage` | `assembleContext()` → `{ system, messages }` | +| **Post-turn hook** | `onChatResponse(result)` | `onChatResponse(result)` (same) | +| **Error handling** | No dedicated hook | `onChatError(error)` | +| **Pre-persist transform** | `sanitizeMessageForPersistence(msg)` | `sanitizeMessageForPersistence(msg)` (same) | +| **Recovery hook** | `onChatRecovery(ctx)` | `onChatRecovery(ctx)` (same) | + +### Storage and data model + +| Concept | AIChatAgent | Think | +| ---------------------- | ----------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------- | +| **Messages** | `this.messages` — mutable field, flat SQL table | `this.messages` — getter from Session tree (always fresh from SQLite) | +| **Storage** | Flat `cf_ai_chat_agent_messages` table | Session: `assistant_messages` (tree with `parent_id`), `assistant_compactions`, `assistant_fts`, `assistant_config` | +| **Regeneration** | Destructive — `_deleteStaleRows` removes old response | Non-destructive — new response branches from same parent, old preserved | +| **Message pruning** | `maxPersistedMessages` (deletes oldest) | Compaction (non-destructive summaries via overlays) | +| **Search** | Not available | FTS5 full-text search (per-session and cross-session) | +| **Context blocks** | Not available | `configureSession()` with writable blocks, skills, search providers | +| **Multi-session** | One conversation per DO | `SessionManager` for multiple conversations per DO | +| **Config persistence** | Not available | `configure(config)` / `getConfig()` with generic `Config` type | + +### Turn execution + +| Concept | AIChatAgent | Think | +| ---------------------- | ------------------------------------------------------------------------- | ----------------------------------------------------------------- | +| **WebSocket chat** | Protocol handler in constructor | Protocol handler via `_setupProtocolHandlers` | +| **Sub-agent RPC** | Not built in | `chat(userMessage, callback, options)` with `StreamCallback` | +| **Programmatic turns** | `saveMessages(messages)` | `saveMessages(messages)` (same) | +| **Continuation** | `continueLastTurn(body?)` — appends to existing message (chunk rewriting) | `continueLastTurn(body?)` — creates new message (append deferred) | +| **Concurrency** | `messageConcurrency` (queue/latest/merge/drop/debounce) | `messageConcurrency` (same strategies, merge is non-destructive) | +| **Durability** | `chatRecovery` + `runFiber` | `chatRecovery` + `runFiber` (same) | +| **Stability** | `waitUntilStable()` / `hasPendingInteraction()` | `waitUntilStable()` / `hasPendingInteraction()` (same) | +| **Turn reset** | `resetTurnState()` (protected) | `resetTurnState()` (protected) | +| **onStart** | Must call `super.onStart()` | Constructor wrapping — no `super.onStart()` needed | + +### Client compatibility + +| Concept | AIChatAgent | Think | +| -------------------------- | ---------------------------------------------------- | ---------------------------------------------------- | +| **useAgentChat** | Primary client hook | Works unchanged (same protocol) | +| **useChat (AI SDK)** | `Response` return type designed for AI SDK internals | `StreamableResult` — works via `useAgentChat` | +| **v4 migration** | `autoTransformMessages` bridges v4→v5 | v5 only (no legacy support) | +| **Message reconciliation** | ID remapping, tool output merge | Session's idempotent append handles underlying cases | +| **Client message sync** | `CF_AGENT_CHAT_MESSAGES` from client | Not needed with Session | +| **Plaintext responses** | Auto-synthesizes UIMessage events | Requires `StreamableResult` | + +--- + +## When to use AIChatAgent + +### 1. You need full control over the LLM call + +You're doing something non-standard — custom streaming, multiple model calls per turn, RAG with vector search before the LLM call, response post-processing, or integrating with a non-AI-SDK provider. AIChatAgent lets you return any `Response` — even a plain text response or a manually constructed SSE stream. + +```typescript +class MyAgent extends AIChatAgent { + async onChatMessage(onFinish, options) { + // Full control: RAG → rerank → generate → post-process + const context = await this.vectorSearch(this.messages); + const response = streamText({ + model: openai("gpt-4o"), + system: buildPrompt(context), + messages: await convertToModelMessages(this.messages), + tools: this.buildTools(), + onFinish + }); + return response.toUIMessageStreamResponse(); + } +} +``` + +### 2. You're migrating from AI SDK v4 + +`autoTransformMessages` handles the v4→v5 format bridge automatically. Think is v5-only — if you have existing v4 clients, AIChatAgent provides the migration path. + +### 3. You want the `Response` abstraction + +If your infrastructure expects HTTP `Response` objects (e.g., for testing, middleware, or non-WebSocket transports), AIChatAgent's `onChatMessage → Response` pattern fits naturally. Think's `StreamableResult` is an internal abstraction. + +### 4. You need message reconciliation + +Multi-tab, multi-device scenarios where clients send optimistic IDs that need server-side remapping. AIChatAgent's `reconcileMessages` handles ID conflicts and tool state merge. Think's Session uses idempotent append which avoids the problem differently — but doesn't remap IDs. + +### 5. You're building a simple chatbot with no memory + +If you don't need context blocks, compaction, search, or multi-session, AIChatAgent is less opinionated — you write your own `onChatMessage` and own exactly the complexity you need. + +--- + +## When to use Think + +### 1. You want to ship fast + +3-line minimal subclass. Override `getModel()` and you have a working chat agent with streaming, persistence, abort/cancel, error handling, and resumable streams. + +```typescript +export class MyAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } +} +``` + +Graduation path: add `getSystemPrompt()` for a custom prompt, `getTools()` for tools, `configureSession()` for memory — each is one method, each has a clear default. + +### 2. You need persistent memory + +Context blocks give the model writable persistent memory via the `set_context` tool. The model can learn and remember facts across conversations without any custom code. + +```typescript +configureSession(session: Session) { + return session + .withContext("memory", { + description: "Important facts about the user.", + maxTokens: 2000 + }) + .withCachedPrompt(); +} +``` + +The memory block content renders into the system prompt with token usage indicators. The model sees: `MEMORY (Important facts — use set_context to update) [42% — 462/1100 tokens]` and can proactively write to it. + +### 3. You need long conversations + +Compaction replaces old messages with LLM-generated summaries — non-destructive, original messages preserved as overlays. Contrast with AIChatAgent's `maxPersistedMessages` which deletes oldest messages (lossy, permanent data loss). + +```typescript +configureSession(session: Session) { + return session + .onCompaction(createCompactFunction({ + summarize: (prompt) => generateText({ model: this.getModel(), prompt }).then(r => r.text) + })) + .compactAfter(50000); +} +``` + +### 4. You need conversation search + +FTS5 full-text search across message history. Per-session `session.search(query)` and cross-session `SessionManager.search(query)`. The model can search its own history via `search_context` tool. + +### 5. You need regeneration with version history + +Think preserves all response alternatives as branches in the Session tree. Users can browse "v1 / v2 / v3" responses via `session.getBranches(messageId)`. AIChatAgent destroys the old response on regeneration. + +### 6. You're building a sub-agent system + +`chat(userMessage, callback)` is designed for parent-child agent communication over Durable Object RPC. The parent drives the child's turns and receives streaming events via `StreamCallback`. + +```typescript +// Parent agent +const child = this.spawn(ChildAgent, "child-1"); +await child.chat("Analyze this data", { + onEvent: (json) => this.forwardToClient(json), + onDone: () => this.handleChildComplete() +}); +``` + +### 7. You need proactive agents + +`saveMessages()` lets the agent inject messages and trigger turns from scheduled tasks, webhooks, or `onChatResponse` hooks — without a WebSocket connection. + +```typescript +async onScheduled() { + await this.saveMessages([{ + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Time for your daily summary." }] + }]); +} +``` + +### 8. You need typed dynamic configuration + +`configure(config)` / `getConfig()` with TypeScript generics. Persisted in Session's `assistant_config` table, survives hibernation and restarts. + +```typescript +class MyAgent extends Think { + async onRequest(request: Request) { + const config = await request.json(); + this.configure(config); + return new Response("OK"); + } +} +``` + +### 9. You need R2-backed skills or on-demand knowledge + +`R2SkillProvider` + `load_context` tool. The model sees skill metadata in the system prompt and loads full content on demand — without bloating the context window. + +--- + +## Architectural advantages Think has over AIChatAgent + +These are structural differences that come from the Session-backed architecture. They can't be added to AIChatAgent without a fundamental storage redesign. + +| Advantage | Why it matters | +| ---------------------------------- | ------------------------------------------------------------------------------ | +| **Tree-structured messages** | Branching, forking, non-destructive regeneration | +| **Context blocks** | Persistent, structured, LLM-writable system prompt sections | +| **Compaction overlays** | Non-destructive summarization — original messages preserved | +| **`assembleContext()` pipeline** | Context blocks → frozen prompt → truncation → pruning, with LLM prefix caching | +| **Session as first-class concept** | Multi-session, cross-session search, usage tracking, forking | +| **`configureSession()` builder** | Discoverable via autocomplete, async-capable, composable | + +## What AIChatAgent has that Think deliberately skips + +| Feature | Rationale | +| ---------------------------------------------------------- | ---------------------------------------------------------------------------- | +| `onFinish` callback on `onChatMessage` | Think uses `onChatResponse` instead — cleaner, fires from all paths | +| `Response` return type | Think uses `StreamableResult` — no HTTP abstraction mismatch | +| v4 → v5 message migration | Think is v5-only — no legacy clients to support | +| `reconcileMessages` | Session's idempotent append + tree structure handles the underlying cases | +| Client message sync (`CF_AGENT_CHAT_MESSAGES` from client) | Unnecessary with Session's tree model | +| `maxPersistedMessages` | Replaced by compaction (non-destructive, preserves information) | +| Plaintext response support | Think requires `StreamableResult` — subclasses can wrap plain text if needed | + +--- + +## Future directions + +### Think could replace AIChatAgent entirely + +Think's `onChatMessage` override gives the same level of control as AIChatAgent — you can ignore all the opinionated defaults and do everything manually. If Think is mature and tested enough, AIChatAgent becomes a legacy API maintained for backward compatibility. + +### Shared Session layer + +AIChatAgent could adopt Session as its storage layer — getting tree messages, compaction, and search without the opinionated framework layer. Session was designed to be reusable. + +### Think-specific client features + +`useAgentChat` works with Think today, but Think-specific features could get first-class client support: + +- **Branch navigation** — `regenerate()` + `getBranches()` for "v1 / v2 / v3" UI +- **Session status** — compaction progress, token usage from `CF_AGENT_SESSION` broadcasts +- **Context block UI** — display memory contents, token budgets +- **Conversation list** — `SessionManager.list()` for conversation sidebars + +### Multi-agent orchestration + +Think's `chat()` RPC + `Session` + `configureSession` make it a natural building block for multi-agent systems where a parent Think agent delegates to child Think agents, each with their own conversation trees, memory, and tools. + +### `StreamableResult` ← `Response` bridge + +Think could accept `Response` objects in `onChatMessage` alongside `StreamableResult`, giving AIChatAgent users a zero-friction migration path. The bridge would parse SSE from the `Response` body into `StreamableResult` chunks. diff --git a/design/think.md b/design/think.md new file mode 100644 index 0000000000..6d7389e479 --- /dev/null +++ b/design/think.md @@ -0,0 +1,504 @@ +# Think + +An opinionated Agent base class for AI assistants. Handles the chat lifecycle — message persistence, agentic loop, streaming, client tools, resumable streams, and extensions — all backed by Durable Object SQLite. + +**Status:** experimental (`@cloudflare/think`, v0.1.2) + +## Problem + +Every AI agent built on the Agents SDK needs the same infrastructure: + +- **Message persistence** — store messages, survive hibernation +- **Streaming** — stream LLM output to clients in real time, handle cancellation +- **Tool execution** — run tools in an agentic loop, manage step limits +- **Error recovery** — persist partial messages on failure, don't lose context +- **Message management** — sanitize provider metadata, enforce storage limits +- **Client tools** — dynamic tool registration from the browser, with result/approval flows +- **Resumable streams** — buffer chunks in SQLite, replay on reconnect + +Building this from scratch for each agent is tedious and error-prone. The base `Agent` class provides the Durable Object primitives (SQLite, WebSocket, RPC, scheduling, fibers) but no opinion on how to run a chat. + +Think is that opinion. + +## Architecture overview + +``` + Browser + | + WebSocket (cf_agent_chat_* protocol) + | + ┌───────┴───────┐ + │ Think │ + │ (top-level) │ + └───────┬───────┘ + | + ┌────────────┼────────────┐ + | | | + SQLite Tables Agentic Loop Tools + (flat messages) (streamText) + | | | + ┌───────┴───────┐ | ┌──────┴──────┐ + │ Messages │ | │ Workspace │ + │ Request Ctx │ | │ Execute │ + │ Config │ | │ Browser │ + └───────────────┘ | │ Extensions │ + | └─────────────┘ +``` + +Think operates in two modes: + +1. **Top-level agent** — speaks the `cf_agent_chat_*` WebSocket protocol directly to browser clients via `useChat` + `AgentChatTransport` +2. **Sub-agent** — called via `chat()` over Durable Object RPC from a parent agent, streaming events through a `StreamCallback` + +Both modes share the same internal lifecycle. The difference is only in how messages arrive and how responses are delivered. + +## How it works + +### Class hierarchy + +``` +Agent (agents SDK — includes runFiber, keepAlive, scheduling, etc.) + └─ Think — adds chat lifecycle, streaming, client tools + └─ YourAgent extends Think — your overrides +``` + +Think extends `Agent` directly. Fiber support (`runFiber`, `stash`, `onFiberRecovered`) is inherited from the base class — no mixin needed. + +### Override points + +Think requires almost no boilerplate. The minimal subclass overrides one method: + +```typescript +export class ChatSession extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } +} +``` + +The full set of override points: + +| Method | Default | Purpose | +| -------------------- | -------------------------------- | ------------------------------------- | +| `getModel()` | throws | Return the `LanguageModel` to use | +| `getSystemPrompt()` | `"You are a helpful assistant."` | System prompt | +| `getTools()` | `{}` | AI SDK `ToolSet` for the agentic loop | +| `getMaxSteps()` | `10` | Max tool-call rounds per turn | +| `assembleContext()` | prune older tool calls | Customize what's sent to the LLM | +| `onChatMessage()` | `streamText(...)` | Full control over inference | +| `onChatError(error)` | passthrough | Customize error handling | + +### Step-by-step: a chat request + +#### 1. Message arrival + +**WebSocket path** (`_handleChatRequest`): + +``` +Client sends: { type: "cf_agent_use_chat_request", id: "req-abc", init: { method: "POST", body: JSON } } +``` + +The body contains `{ messages: UIMessage[], clientTools?: ClientToolSchema[] }`. Think appends each incoming message via `INSERT OR IGNORE` (idempotent on message ID), then reloads the full message list from SQLite. Client tool schemas are captured and persisted to SQLite (`think_request_context`) so they survive hibernation. + +**RPC path** (`chat()`): + +```typescript +await session.chat("Summarize the project", callback, { signal, tools }); +``` + +The parent agent calls `chat()` directly with a string or `UIMessage`. + +#### 2. Abort controller setup + +Each WebSocket request gets its own `AbortController`, keyed by request ID. The controller's signal is threaded through `onChatMessage()` → `streamText()` → the LLM provider. The `cf_agent_chat_request_cancel` message triggers `controller.abort()`. + +For the RPC path, the caller passes an `AbortSignal` via `ChatOptions`. + +#### 3. Agentic loop (`onChatMessage`) + +The default implementation calls the AI SDK's `streamText()`: + +```typescript +streamText({ + model: this.getModel(), + system: this.getSystemPrompt(), + messages: await this.assembleContext(), + tools: { ...this.getTools(), ...clientToolSet, ...options?.tools }, + stopWhen: stepCountIs(this.getMaxSteps()), + abortSignal: options?.signal +}); +``` + +Client tool schemas (from the browser) are merged into the tool set via `createToolsFromClientSchemas()`. RPC-provided tools are also merged. + +The agentic loop runs until: + +- The model produces a text response with no tool calls (natural completion) +- The step count limit is reached +- The abort signal fires (user cancelled) +- An error occurs + +#### 4. Context assembly (`assembleContext`) + +The default implementation converts `this.messages` (UIMessage format) to model messages and prunes old tool calls: + +```typescript +pruneMessages({ + messages: await convertToModelMessages(this.messages), + toolCalls: "before-last-2-messages" +}); +``` + +Override this to inject memory, project context, RAG results, or compaction summaries. + +#### 5. Streaming + +**WebSocket path** (`_streamResult`): + +The `streamText()` result is iterated via `toUIMessageStream()`. Each chunk is simultaneously: + +- **Applied to a `StreamAccumulator`** — builds the assistant `UIMessage` incrementally (text parts, reasoning, tool calls, tool results, sources, files). The accumulator detects error chunks and cross-message tool updates. +- **Stored for resumability** — `ResumableStream.storeChunk()` buffers chunks in SQLite for replay on reconnect. +- **Broadcast to clients** — each chunk is sent as `{ type: "cf_agent_use_chat_response", id, body: JSON, done: false }`, excluding connections pending stream resume. + +When the stream completes: + +``` +{ type: "cf_agent_use_chat_response", id, body: "", done: true } +``` + +**RPC path** (`chat`): + +Uses a separate `StreamAccumulator` and calls `callback.onEvent(json)` for each chunk, `callback.onDone()` on completion, `callback.onError(msg)` on error. + +#### 6. Persistence + +After the stream completes, the assembled assistant message is persisted with three transformations: + +1. **Sanitize** — `sanitizeMessage()` strips provider ephemeral metadata (`itemId`, `reasoningEncryptedContent`), removes empty reasoning parts +2. **Enforce row size** — `enforceRowSizeLimit()` compacts tool outputs exceeding 1.8 MB (SQLite has a ~2 MB row limit) +3. **Incremental persist** — compares the serialized message to `_persistedMessageCache`. If unchanged, skips the SQL write. Uses `INSERT ON CONFLICT DO UPDATE` for the upsert. + +After persistence, `maxPersistedMessages` is enforced by counting all messages and deleting the oldest ones beyond the limit. The updated message list is broadcast to all clients. + +A `_turnQueue.generation` check prevents persisting into a cleared conversation — if the user cleared the chat while streaming, the generation counter will have changed and persistence is skipped. + +#### 7. Error handling + +If an error occurs during the agentic loop or streaming: + +- **Partial message is persisted** — whatever was generated before the error is saved so context isn't lost (both WebSocket and RPC paths) +- **`onChatError(error)` is called** — override to log, transform, or swallow +- **Error is communicated** — WebSocket broadcasts `{ done: true, error: true }`, RPC calls `callback.onError()` + +### Wire protocol + +Think speaks the same WebSocket protocol as `@cloudflare/ai-chat`, making it compatible with `useAgentChat` and `useChat` + `AgentChatTransport`. + +| Direction | Message type | Purpose | +| --------------- | -------------------------------- | ----------------------------------------------------------- | +| Client → Server | `cf_agent_use_chat_request` | Send a chat message (contains `{ messages, clientTools? }`) | +| Client → Server | `cf_agent_chat_clear` | Clear the current conversation | +| Client → Server | `cf_agent_chat_request_cancel` | Cancel a specific request by ID | +| Client → Server | `cf_agent_tool_result` | Client tool result (output, state, optional error) | +| Client → Server | `cf_agent_tool_approval` | Tool approval/denial response | +| Client → Server | `cf_agent_stream_resume_request` | Request stream replay after reconnect | +| Client → Server | `cf_agent_stream_resume_ack` | Acknowledge stream resume, trigger chunk replay | +| Server → Client | `cf_agent_use_chat_response` | Stream chunk (`done: false`) or completion (`done: true`) | +| Server → Client | `cf_agent_chat_messages` | Full message list broadcast (after persistence) | +| Server → Client | `cf_agent_chat_clear` | Confirm conversation was cleared | +| Server → Client | `cf_agent_stream_resuming` | Notify client that a stream is active and can be resumed | +| Server → Client | `cf_agent_stream_resume_none` | No active stream to resume | +| Server → Client | `cf_agent_message_updated` | Single message update (after tool result/approval applied) | + +### Client tools + +Client tools are tools defined by the browser at runtime (via `clientTools` in the chat request body). Think handles the full lifecycle: + +1. **Registration** — client sends `ClientToolSchema[]` with the chat request. Think converts them to AI SDK tools via `createToolsFromClientSchemas()` and merges them into the tool set. + +2. **Schema persistence** — `_lastClientTools` is persisted to `think_request_context` (SQLite) so client tools survive hibernation and are available during auto-continuations. + +3. **Tool result** — client sends `cf_agent_tool_result` with `{ toolCallId, output, state?, errorText?, autoContinue?, clientTools? }`. Think finds the matching tool part in `this.messages`, updates its state to `output-available` (or `output-error`), persists the updated message, and broadcasts `cf_agent_message_updated`. + +4. **Tool approval** — client sends `cf_agent_tool_approval` with `{ toolCallId, approved, autoContinue? }`. Think updates the tool part state to `approval-responded` (if approved) or `output-denied` (if denied), persists, and broadcasts. + +5. **Auto-continuation** — when `autoContinue: true` is set on a tool result or approval, Think schedules a continuation turn after a 50ms coalesce window. This batches rapid-fire tool results into a single LLM call. The continuation runs the full `onChatMessage()` → stream → persist pipeline. Deferred continuations queue up if a continuation is already in flight. + +### Resumable streaming + +Think uses `ResumableStream` from `agents/chat` for stream resumability: + +1. **Chunk buffering** — during streaming, each chunk is stored in SQLite via `ResumableStream.storeChunk()`. + +2. **Reconnect detection** — when a client connects (`onConnect`), Think checks for an active stream and sends `cf_agent_stream_resuming`. The client is added to `_pendingResumeConnections` and excluded from live chunk broadcasts to avoid duplicates. + +3. **Replay** — when the client sends `cf_agent_stream_resume_ack`, Think replays all buffered chunks via `ResumableStream.replayChunks()`. If the stream was orphaned (restored from SQLite after hibernation with no live reader), the partial assistant message is reconstructed from chunks and persisted. + +4. **Continuation coordination** — `ContinuationState` tracks pending, active, and deferred continuation requests. Connections awaiting a continuation stream to start are queued and notified when the stream begins. + +### Message storage + +Think uses a flat `assistant_messages` table — no tree structure, no branching, no sessions: + +```sql +CREATE TABLE assistant_messages ( + id TEXT PRIMARY KEY, + role TEXT NOT NULL, + content TEXT NOT NULL, -- JSON-serialized UIMessage + created_at DATETIME DEFAULT CURRENT_TIMESTAMP +) +``` + +Messages are ordered by `created_at` on load. User messages use `INSERT OR IGNORE` (idempotent). Assistant messages use `INSERT ON CONFLICT DO UPDATE` (streaming builds incrementally). + +A separate table stores request context across hibernation: + +```sql +CREATE TABLE think_request_context ( + key TEXT PRIMARY KEY, + value TEXT NOT NULL +) +``` + +Currently stores only `lastClientTools`. + +### Dynamic configuration + +Think accepts a `Config` type parameter for per-instance configuration: + +```typescript +export class ChatSession extends Think { + getModel() { + const tier = this.getConfig()?.modelTier ?? "fast"; + return MODELS[tier]; + } +} +``` + +Configuration is stored in SQLite (`_think_config`) and cached in memory. It survives hibernation. A parent orchestrator can configure sub-agents via RPC: + +```typescript +const session = await this.subAgent(ChatSession, "agent-abc"); +await session.configure({ modelTier: "capable" }); +``` + +### Sub-agent RPC entry point + +When used as a sub-agent, the `chat()` method runs a full turn and streams events via a callback: + +```typescript +interface StreamCallback { + onEvent(json: string): void | Promise; + onDone(): void | Promise; + onError?(error: string): void | Promise; +} +``` + +The parent implements `StreamCallback` as an `RpcTarget` (so it crosses the DO RPC boundary). The `chat()` method handles the full lifecycle: persist user message, call `onChatMessage()`, iterate stream, persist assistant message, handle errors. + +### Clear + +Clearing (`cf_agent_chat_clear`) is comprehensive: + +1. Reset the turn queue (increments generation, invalidating queued turns) +2. Abort all in-flight requests +3. Clear resumable stream state +4. Clear continuation state (pending, deferred, awaiting connections) +5. Clear client tools +6. Delete all messages from SQLite +7. Clear in-memory message list and persistence cache +8. Broadcast `cf_agent_chat_clear` to all clients + +### Durable fibers + +Think inherits `runFiber()` from the `Agent` base class. Fiber state is persisted in `cf_agents_runs` (SQLite). See [forever.md](../experimental/forever.md) for the full design. + +**Note:** Think does not currently wire fibers into the chat lifecycle. There is no `chatRecovery` flag and no `onChatRecovery` hook. Chat turns are not wrapped in `runFiber` — they rely on `keepAliveWhile()` to prevent eviction during streaming. + +## Tools + +Think provides a built-in workspace and factory functions for additional tool patterns. + +### Built-in workspace + +Every Think instance gets `this.workspace` — a `Workspace` (from `@cloudflare/shell`) backed by the DO's SQLite storage. Workspace tools (`read`, `write`, `edit`, `list`, `find`, `grep`, `delete`) are automatically merged into every `onChatMessage` call, before `getTools()`. + +Override to add R2 spillover: `override workspace = new Workspace({ sql: this.ctx.storage.sql, r2: this.env.R2, name: () => this.name })`. + +### Workspace tools (`@cloudflare/think/tools/workspace`) + +The individual tool factories are also exported for custom storage backends. Seven file operation tools backed by abstract operation interfaces (`ReadOperations`, `WriteOperations`, etc.). + +| Tool | Description | Operations interface | +| ---------------- | ------------------------------------------------- | -------------------- | +| `read_file` | Read file contents | `ReadOperations` | +| `write_file` | Create or overwrite a file | `WriteOperations` | +| `edit_file` | Find-and-replace edit (rejects ambiguous matches) | `EditOperations` | +| `list_directory` | List directory contents with metadata | `ListOperations` | +| `find_files` | Glob pattern search | `FindOperations` | +| `grep` | Regex search across files | `GrepOperations` | +| `delete` | Delete files or directories | `DeleteOperations` | + +All tools use Zod v4 schemas for input validation. + +### Code execution (`@cloudflare/think/tools/execute`) + +A sandboxed JavaScript execution tool powered by `@cloudflare/codemode`: + +```typescript +const executeTool = createExecuteTool({ + tools: workspaceTools, // available as codemode.* in sandbox + state: workspaceBackend, // optional: available as state.* in sandbox + providers: [], // optional: additional named namespaces + loader: this.env.LOADER +}); +``` + +The LLM writes JavaScript code. The tool sends it to a dynamic Worker isolate via `DynamicWorkerExecutor`. The sandbox can call workspace tools via `codemode.*` and optionally the full `state.*` filesystem API (`readFile`, `writeFile`, `glob`, `searchFiles`, `planEdits`, etc.). Fully isolated: no network access by default, configurable timeout. + +### Browser tools (`@cloudflare/think/tools/browser`) + +Two AI SDK tools for CDP-based browser automation: + +- **`browser_search`** — query the CDP protocol spec to discover commands, events, and types. The model writes JavaScript that runs against a normalized copy of the protocol, exposed via `spec.get()`. +- **`browser_execute`** — run CDP commands against a live browser session. The model writes JavaScript that calls `cdp.send()`, `cdp.attachToTarget()`, and debug log helpers. + +Both tools delegate to `createBrowserToolHandlers` from `agents/browser`, reusing the same code-mode sandbox and CDP session management. Requires a Browser Rendering binding (`browser`) and a `WorkerLoader` (`loader`). + +```typescript +createBrowserTools({ + browser: this.env.BROWSER, + loader: this.env.LOADER +}); +``` + +### Extensions (`@cloudflare/think/tools/extensions`) + +Two AI SDK tools for managing extensions at runtime: + +- **`load_extension`** — LLM writes a JS object expression defining tools, Think loads it as a sandboxed Worker via `WorkerLoader` +- **`list_extensions`** — lists currently loaded extensions and their tools + +### Extension system (`@cloudflare/think/extensions`) + +`ExtensionManager` handles the full extension lifecycle: + +1. **Loading** — wraps extension source in a Worker module with `describe()` / `execute()` RPC, loads via `WorkerLoader` with permission-gated bindings +2. **Tool discovery** — calls `describe()` to get tool descriptors (JSON Schema inputs), exposes as AI SDK tools with namespaced names (`{extensionName}_{toolName}`) +3. **Persistence** — stores extension manifest + source in DO storage, `restore()` rebuilds from storage after hibernation +4. **Permissions** — extensions declare `network` (allowed hosts) and `workspace` (`read` | `read-write` | `none`) permissions. Workspace access is mediated by `HostBridgeLoopback`, a `WorkerEntrypoint` that resolves the parent agent via `ctx.exports` and delegates operations with permission checks. +5. **Unloading** — removes the extension and its tools, deletes from storage + +## SQLite tables + +| Table | Owner | Purpose | +| ----------------------- | ----------------- | ---------------------------------------------- | +| `assistant_messages` | Think | Flat ordered message store (no branching) | +| `think_request_context` | Think | Client tools and request context (hibernation) | +| `_think_config` | Think | Dynamic configuration (key-value) | +| `cf_agents_runs` | Agent (inherited) | Durable fiber state and checkpoints | +| `cf_agents_schedules` | Agent (inherited) | Scheduled tasks and intervals | + +## Known gaps (vs AIChatAgent) + +Features present in `@cloudflare/ai-chat` but not yet in Think: + +| Feature | AIChatAgent | Think | +| ------------------------------------ | ------------------------------------------------ | -------------------------------------------------- | +| Multi-session / branching | No | No (flat table, no session ID) | +| `saveMessages()` | Programmatic message injection + turn trigger | Not implemented | +| `continueLastTurn()` | Continue from last assistant message | Not implemented | +| `chatRecovery` / `onChatRecovery` | Fiber-wrapped turns, recovery after eviction | Not implemented (has fibers but not wired to chat) | +| `onChatResponse` hook | Post-turn lifecycle callback | Not implemented | +| `onSanitizeMessage` hook | Custom message transformation before persistence | Not implemented | +| `waitUntilStable()` | Await conversation quiescence | Not implemented | +| `hasPendingInteraction()` | Track pending client tool state | Not implemented | +| Message reconciliation | ID remapping, dedup, merge on client sync | `INSERT OR IGNORE` only | +| Regeneration | `regenerate-message` trigger | Not implemented | +| `messageConcurrency` strategies | queue, latest, merge, drop, debounce | Queue only (via TurnQueue) | +| Custom body persistence | `_lastBody` persisted to SQLite | Not parsed or persisted | +| `CF_AGENT_CHAT_MESSAGES` from client | Full array sync from client | Not handled | +| `onFinish` callback | Provider-level finish metadata | Not exposed | +| v4 → v5 message migration | `autoTransformMessages()` | Not implemented (v5 only) | +| Compaction | No (only in experimental Session) | Not implemented | +| Context blocks | No (only in experimental Session) | Not implemented | + +## Key decisions + +### Why a base class instead of a mixin? + +Think is more than a behavior addition — it's an opinion about how chat agents work. The message store, streaming protocol, persistence pipeline, and error handling are deeply intertwined. A mixin would force awkward composition with other mixins that might conflict on `onMessage`, `onStart`, or storage tables. A base class makes the lifecycle explicit and predictable. + +### Why `StreamAccumulator` instead of inline chunk parsing? + +AIChatAgent uses `applyChunkToParts()` with manual state tracking. Think uses `StreamAccumulator` (from `agents/chat`) which encapsulates the same logic behind a cleaner interface — `applyChunk()` returns a `ChunkResult` with optional actions (cross-message tool updates, errors). This avoids duplicating the chunk-to-parts logic. + +### Why INSERT OR IGNORE for user messages, INSERT ON CONFLICT UPDATE for assistant messages? + +User messages arrive from the client with stable IDs. The same message may arrive multiple times (reconnect, retry). `INSERT OR IGNORE` makes this idempotent. + +Assistant messages are built incrementally during streaming. The first persist inserts; subsequent persists need to update the content. `INSERT ON CONFLICT DO UPDATE` handles both cases. + +### Why a persistence cache? + +The `_persistedMessageCache` maps message IDs to their last-persisted JSON. Before writing to SQLite, Think compares the current serialization to the cached version. If identical, the write is skipped. Without the cache, every broadcast would trigger unnecessary SQL writes. + +### Why sanitize messages before persistence? + +LLM providers attach ephemeral metadata to messages (OpenAI's `itemId`, `reasoningEncryptedContent`). This metadata is meaningless after the response is complete and wastes storage. Sanitization strips it before persistence. + +### Why enforce row size limits? + +Durable Object SQLite has a ~2 MB row size limit. Tool outputs (especially from code execution or file reads) can easily exceed this. Rather than failing the entire persistence operation, Think truncates oversized parts with a clear marker. The threshold is 1.8 MB, leaving headroom. + +### Why the loopback pattern for extensions? + +Extension Workers loaded via `WorkerLoader` can only receive `Fetcher`/`ServiceStub` in their `env`, not `RpcStub`. The `HostBridgeLoopback` is a `WorkerEntrypoint` that carries serializable props and resolves the actual agent at call time via `ctx.exports`. See [loopback.md](./loopback.md). + +## Tradeoffs + +**Think is opinionated.** It assumes UIMessage format, the AI SDK's `streamText` interface, and a specific WebSocket protocol. Agents that need a fundamentally different message format or streaming protocol should use the base `Agent` class directly. + +**All messages in memory.** `this.messages` holds the full conversation. For very long conversations, this could be expensive. `maxPersistedMessages` is a partial mitigation. Compaction is not yet implemented. + +**Single conversation per instance.** Think currently stores all messages in a single flat table with no session ID. There is no multi-session support. The `SessionManager` from `agents/experimental/memory/session` is designed to fill this gap but has not been integrated. + +**No message reconciliation.** Think uses `INSERT OR IGNORE` for incoming messages — it does not handle the client sending edited or truncated message lists. Regeneration (re-running from an earlier point) is not supported. + +**Extension sandbox is all-or-nothing on network.** The `permissions.network` field declares allowed hosts, but actual enforcement is binary: either no network or full network. Per-host filtering is not yet implemented at the runtime level. + +## Testing + +Tests in `packages/think/src/tests/`, running inside the Workers runtime via `@cloudflare/vitest-pool-workers`: + +- **Core chat** (`think-session.test.ts`) — send, multi-turn, persistence, streaming, clear, UIMessage input, getMessages +- **Error handling** (`think-session.test.ts`) — error messages, partial persistence, error hooks, recovery after error +- **Abort** (`think-session.test.ts`) — stop streaming, persist partial on abort, callback not called after abort +- **Agentic loop** (`assistant-agent-loop.test.ts`) — text-only, with tools, context assembly, model errors, custom getTools +- **WebSocket protocol** (`assistant-agent.test.ts`) — send, stream, persistence via WS, clear, resumable streaming +- **Client tools** (`client-tools.test.ts`) — tool result application, tool approval, auto-continuation, schema persistence +- **Extensions** (`extension-manager.test.ts`) — load, unload, restore, tool creation, permissions, namespacing +- **Fibers** (`fiber.test.ts`) — runFiber execution, checkpoint via ctx.stash, fire-and-forget, recovery via onFiberRecovered +- **Tools** (`assistant-tools.test.ts`) — workspace tools, code execution tool +- **E2E** (`assistant-e2e.test.ts`) — end-to-end WebSocket flows + +## Package exports + +| Import path | Source | Purpose | +| ------------------------------------ | ------------------------- | ------------------------------------------------------ | +| `@cloudflare/think` | `src/think.ts` | Think base class, Session, Workspace re-exports, types | +| `@cloudflare/think/extensions` | `src/extensions/index.ts` | ExtensionManager, HostBridgeLoopback | +| `@cloudflare/think/tools/workspace` | `src/tools/workspace.ts` | File operation tool factories (for custom backends) | +| `@cloudflare/think/tools/execute` | `src/tools/execute.ts` | Sandboxed code execution tool | +| `@cloudflare/think/tools/browser` | `src/tools/browser.ts` | CDP browser automation tools (search + execute) | +| `@cloudflare/think/tools/extensions` | `src/tools/extensions.ts` | Extension management AI tools | + +## History + +- [chat-shared-layer.md](./chat-shared-layer.md) — shared streaming, sanitization, and protocol primitives (Think uses `StreamAccumulator`, `sanitizeMessage`, `enforceRowSizeLimit`, `CHAT_MESSAGE_TYPES`, `TurnQueue`, `ResumableStream`, `ContinuationState` from `agents/chat`) +- [rfc-sub-agents.md](./rfc-sub-agents.md) — sub-agents via facets (Think's `subAgent()` is built on this) +- [loopback.md](./loopback.md) — cross-boundary RPC pattern (used by extension host bridge) +- [workspace.md](./workspace.md) — Workspace design (Think's file tools are backed by this) diff --git a/design/visuals.md b/design/visuals.md new file mode 100644 index 0000000000..414a663074 --- /dev/null +++ b/design/visuals.md @@ -0,0 +1,113 @@ +# Visuals & UI + +## Design system: Kumo + +The playground (and eventually all examples) uses [Kumo](https://kumo-ui.com/), Cloudflare's internal design system (`@cloudflare/kumo`). It gives us semantic color tokens, accessible components, and automatic light/dark mode — all without maintaining our own component primitives. + +### Setup + +- **Package**: `@cloudflare/kumo` (installed at monorepo root as a devDependency) +- **Icons**: `@phosphor-icons/react` v2 (Kumo's peer icon library, also at root). Always use the `*Icon` suffixed exports (e.g. `TrashIcon`, `ShieldIcon`) — the bare names (`Trash`, `Shield`) are deprecated. +- **Tailwind v4**: Requires `@tailwindcss/vite` in `vite.config.ts` (alongside `@vitejs/plugin-react` and `@cloudflare/vite-plugin`). Kumo ships its own Tailwind plugin; imported in `styles.css`: + ```css + @source "../../../node_modules/@cloudflare/kumo/dist/**/*.{js,jsx,ts,tsx}"; + @import "tailwindcss"; + @import "@cloudflare/kumo/styles/tailwind"; + ``` + Note: the `@source` path must point to the hoisted Kumo package at the monorepo root (`../../../node_modules`), not a local `node_modules`. + +### Dark mode + +Kumo uses a `data-mode` attribute on `` (not Tailwind's `dark:` class variant). Each example includes an inline `ModeToggle` component that sets `document.documentElement.setAttribute("data-mode", mode)` via `useState`/`useEffect` and persists to `localStorage`. All Kumo semantic tokens (`bg-kumo-base`, `text-kumo-default`, `border-kumo-line`, etc.) respond to this automatically — no `dark:` prefixes anywhere in the codebase. + +### Color themes + +All examples use Kumo's default theme — no custom theme overrides. Kumo supports theming via a `data-theme` attribute on a parent element if needed in the future, but current examples omit it. + +### Standard UI patterns + +Each example includes these UI elements, inlined per-example rather than from a shared package: + +| Pattern | Source | Purpose | +| --------------------- | ------------------- | --------------------------------------------------------------------------------- | +| `PoweredByCloudflare` | `@cloudflare/kumo` | "Powered by Cloudflare" footer badge — **every example should include this** | +| `CloudflareLogo` | `@cloudflare/kumo` | Cloudflare logo component with glyph/full variants | +| `ModeToggle` | Inlined per example | Light/dark toggle using `localStorage` + `data-mode` attribute | +| `ConnectionIndicator` | Inlined per example | Colored dot + label for WebSocket state (`connecting`/`connected`/`disconnected`) | + +Dark mode is managed via the `data-mode` attribute on ``. The `index.html` flash-prevention script reads from `localStorage` on page load; the inlined `ModeToggle` component handles runtime toggling. + +### Routing integration + +Kumo's `` lets you inject a custom link component so `` renders via your router. However, there's a type mismatch between Kumo and React Router that requires an adapter. + +**The problem:** Kumo's `LinkComponentProps` defines `to?: string` (optional), but React Router's `Link` requires `to: To` (non-optional, and `To = string | Partial`). These types aren't assignable in either direction — you can't pass `RouterLink` directly to `LinkProvider` without a type error. + +**Our fix:** A thin `AppLink` adapter in `client.tsx` that bridges the two: + +```tsx +const AppLink = forwardRef( + ({ to, ...props }, ref) => { + if (to) { + return ; + } + return ; + } +); +``` + +This handles the optionality gap (falls back to `` when `to` is absent) and narrows `To` to `string` which is all Kumo ever passes. + +**Upstream fix:** Either Kumo should make `to` required in `LinkComponentProps` (it's always provided when the component is actually called), or accept `To` from React Router. Alternatively, React Router could loosen `Link` to accept `to?: string`. Worth raising with the Kumo team since every React Router user will hit this. + +## Kumo components we use + +| Kumo component | Replaces | +| ----------------------- | ----------------------------------------------------------------------------------------------------- | +| `Button` | All buttons (primary, secondary, destructive, ghost actions) | +| `Input` | Text inputs; uses built-in `label` prop for Field wrapper | +| `InputArea` | Textareas | +| `Surface` | Card/panel containers | +| `Text` | Headings and body text (note: does **not** accept `className` — wrap in a `
` for margin/spacing) | +| `Badge` | Status indicators and tags | +| `Banner` | Alert/warning banners | +| `CodeBlock` | Static code examples and dynamic JSON display | +| `Tabs` | Tab switchers (e.g. inbox/outbox) | +| `Switch` | Boolean toggles (with built-in label) | +| `Checkbox` | Multi-select checkboxes | +| `Table` | Data tables | +| `Empty` | Empty-state placeholders | +| `Loader` | Loading spinners | +| `LinkProvider` / `Link` | Router-aware links | + +## What we do custom (and why) + +### Sidebar category toggle + +The sidebar nav uses a raw ` + +
+ ); +} +``` + +### Vanilla JavaScript + +```typescript +import { AgentClient } from "agents/client"; + +const agent = new AgentClient({ + agent: "Counter", + name: "user-123", // Optional: unique instance name + host: window.location.host, + onStateUpdate: (state) => { + // Update the DOM when state changes + document.getElementById("count").textContent = String(state.count); + } +}); + +// Call methods — agent.state is also readable directly +document.getElementById("increment").onclick = () => agent.call("increment"); +``` + +--- + +## Adding Multiple Agents + +Add more agents by extending the configuration: + +```typescript +// src/agents/chat.ts +export class Chat extends Agent { + // ... +} + +// src/agents/scheduler.ts +export class Scheduler extends Agent { + // ... +} +``` + +Update `wrangler.jsonc`: + +```jsonc +{ + "durable_objects": { + "bindings": [ + { "name": "Counter", "class_name": "Counter" }, + { "name": "Chat", "class_name": "Chat" }, + { "name": "Scheduler", "class_name": "Scheduler" } + ] + }, + "migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["Counter", "Chat", "Scheduler"] + } + ] +} +``` + +Export all agents from your entry point: + +```typescript +export { Counter } from "./agents/counter"; +export { Chat } from "./agents/chat"; +export { Scheduler } from "./agents/scheduler"; +``` + +--- + +## Common Integration Patterns + +### Agents Behind Authentication + +Check auth before routing to agents: + +```typescript +export default { + async fetch(request: Request, env: Env) { + // Check auth for agent routes + if (request.url.includes("/agents/")) { + const authResult = await checkAuth(request, env); + if (!authResult.valid) { + return new Response("Unauthorized", { status: 401 }); + } + } + + const agentResponse = await routeAgentRequest(request, env); + if (agentResponse) return agentResponse; + + // ... rest of routing + } +}; +``` + +### Custom Agent Path Prefix + +By default, agents are routed at `/agents/{agent-name}/{instance-name}`. You can customize this: + +```typescript +import { routeAgentRequest } from "agents"; + +const agentResponse = await routeAgentRequest(request, env, { + prefix: "/api/agents" // Now routes at /api/agents/{agent-name}/{instance-name} +}); +``` + +### Accessing Agents from Server Code + +You can interact with agents directly from your Worker code: + +```typescript +import { getAgentByName } from "agents"; + +export default { + async fetch(request: Request, env: Env) { + if (request.url.endsWith("/api/increment")) { + // Get a specific agent instance + const counter = await getAgentByName(env.Counter, "shared-counter"); + const newCount = await counter.increment(); + return Response.json({ count: newCount }); + } + // ... + } +}; +``` + +--- + +## Troubleshooting + +### "Agent not found" or 404 errors + +1. **Check the export** - Agent class must be exported from your main entry point +2. **Check the binding** - `class_name` in `wrangler.jsonc` must match the exported class name exactly +3. **Check the route** - Default route is `/agents/{agent-name}/{instance-name}` + +### "No such Durable Object class" error + +Add the migration to `wrangler.jsonc`: + +```jsonc +"migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["YourAgentClass"] + } +] +``` + +### WebSocket connection fails + +Ensure your routing passes the response through unchanged: + +```typescript +// ✅ Correct - return the response directly +const agentResponse = await routeAgentRequest(request, env); +if (agentResponse) return agentResponse; + +// ❌ Wrong - don't wrap or modify the response +const agentResponse = await routeAgentRequest(request, env); +if (agentResponse) return new Response(agentResponse.body); // Breaks WebSocket +``` + +### State not persisting + +Check that: + +1. You're using `this.setState()`, not mutating `this.state` directly +2. The agent class is in `new_sqlite_classes` in migrations +3. You're connecting to the same agent instance name + +--- + +## Next Steps + +- [State Management](./state.md) - Deep dive into agent state +- [Scheduling](./scheduling.md) - Background tasks and cron jobs +- [Agent Class](./agent-class.md) - Full lifecycle and methods +- [Client SDK](./client-sdk.md) - Complete client API reference diff --git a/docs/agent-class.md b/docs/agent-class.md new file mode 100644 index 0000000000..9b826895fd --- /dev/null +++ b/docs/agent-class.md @@ -0,0 +1,458 @@ +# Demystifying the Agent class + +The core of the `agents` library is the exported `Agent` class. Following the pattern from [Durable Objects](https://developers.cloudflare.com/durable-objects/api/), the main API for developers is to extend the `Agent` so those classes inherit all the built-in features. While this effectively is a supercharged primitive that allows developers to only write the logic they need in their agents, it obscures the inner workings. + +This document tries to bridge that gap, empowering any developer aiming to get started writing agents to get the full picture and avoid common pitfalls. The snippets shown here are primarily illustrative and don't necessarily represent best practices. For a more in-depth look at the inner workings of the `Agent` class, check out the [API reference](https://developers.cloudflare.com/agents/api-reference/) and the [source code](https://github.com/cloudflare/agents/blob/main/packages/agents/src/index.ts). + +# What is the Agent? + +The `Agent` class is an extension of `DurableObject`. That is to say, they _are_ **Durable Objects**. If you're not familiar with Durable Objects, it is highly recommended that you read ["What are Durable Objects"](https://developers.cloudflare.com/durable-objects/) but at their core, Durable Objects are globally addressable (each instance has a unique ID) single-threaded compute instances with long term storage (KV/SQLite). +That being said, `Agent` does **not** extend `DurableObject` directly but instead `Server`. `Server` is a class provided by [PartyKit](https://github.com/cloudflare/partykit/tree/main/packages/partyserver). + +You can visualize the logic as a Matryoshka doll: **DurableObject** -> **Server** -> **Agent**. + +## Layer 0: Durable Object + +This won't cover Durable Objects in detail, but it's good to know what primitives they expose so we understand how the outer layers make use of them. The Durable Object class comes with: + +### `constructor` + +```ts +constructor(ctx: DurableObjectState, env: Env) {} +``` + +The Workers runtime always calls the constructor to handle things internally. This means 2 things: + +1. While the constructor is called every time the DO is initialized, the signature is fixed. Developers **can't add or update parameters from the constructor**. +2. Instead of instantiating the class manually, developers must use the binding APIs and do it through the [DurableObjectNamespace](https://developers.cloudflare.com/durable-objects/api/namespace/). + +### RPC + +By writing a Durable Object class which inherits from the built-in type `DurableObject`, public methods are exposed as RPC methods, which developers can call using a [DurableObjectStub from a Worker](https://developers.cloudflare.com/durable-objects/best-practices/create-durable-object-stubs-and-send-requests/#invoking-methods-on-a-durable-object). + +```ts +// This instance could've been active, hibernated, +// not initialized or maybe had never even been created! +const stub = env.MY_DO.getByName("foo"); + +// We can call any public method of the class since. The runtime +// **ensures** the constructor is called for us if the instance wasn't active. +await stub.bar(); +``` + +### `fetch()` + +Durable Objects can take a `Request` from a Worker and send a `Response` back. This can **only** be done through the [`fetch`](https://developers.cloudflare.com/durable-objects/best-practices/create-durable-object-stubs-and-send-requests/#invoking-the-fetch-handler) method (which the developer must implement). + +### WebSockets + +Durable Objects include first-class support for [WebSockets](https://developers.cloudflare.com/durable-objects/best-practices/websockets/). A DO can accept a WebSocket it receives from a `Request` in `fetch` and forget about it. The base class provides methods that developers can implement that are called as callbacks. They effectively replace the need for event listeners. + +The base class provides `webSocketMessage(ws, message)`, `webSocketClose(ws, code, reason, wasClean)` and `webSocketError(ws , error)` ([API](https://developers.cloudflare.com/workers/runtime-apis/websockets)). + +```ts +export class MyDurableObject extends DurableObject { + async fetch(request) { + // Creates two ends of a WebSocket connection. + const webSocketPair = new WebSocketPair(); + const [client, server] = Object.values(webSocketPair); + + // Calling `acceptWebSocket()` connects the WebSocket to the Durable Object, allowing the WebSocket to send and receive messages. + this.ctx.acceptWebSocket(server); + + return new Response(null, { + status: 101, + webSocket: client + }); + } + + async webSocketMessage(ws, message) { + // echo back the messages + ws.send(msg); + } +} +``` + +### `alarm()` + +HTTP and RPC requests are not the only entrypoints for a DO. Alarms allow developers to schedule an event to trigger at a later time. Whenever the next alarm is due, the runtime will call the `alarm()` method, which is left to the developer to implement. + +To schedule an alarm, you can use the `this.ctx.storage.setAlarm()` method. For more information, check [the documentation](https://developers.cloudflare.com/durable-objects/api/alarms/). + +### `this.ctx` + +The base `DurableObject` class sets the [DurableObjectState](https://developers.cloudflare.com/durable-objects/api/state/) into `this.ctx`. There are a lot of interesting methods and properties, but we'll focus on `this.ctx.storage`. + +### `this.ctx.storage` + +[DurableObjectStorage](https://developers.cloudflare.com/durable-objects/api/sqlite-storage-api/) is the main interface with the DO's persistence mechanisms, which include both a KV and SQLITE **synchronous** APIs. + +```ts +const sql = this.ctx.storage.sql; +const kv = this.ctx.storage.kv; + +// An example of a synchronous SQL query +const rows = sql.exec("SELECT * FROM contacts WHERE country = ?", "US"); + +// And an example of the synchronous KV +const token = kv.get("someToken"); +``` + +### `this.ctx.env` + +Lastly, it's worth mentioning that the DO also has the Worker `Env` in `this.env`. Read more [here](https://developers.cloudflare.com/workers/runtime-apis/bindings). + +## Layer 1: Partykit `Server` + +Now that you've seen what Durable Objects come with out-of-the-box, what [PartyKit](https://github.com/cloudflare/partykit)'s `Server` (package `partyserver`) implements will be clearer. It's an **opinionated `DurableObject` wrapper that improves DX by hiding away DO primitives in favor of more developer friendly callbacks**. + +An important note is that `Server` **does NOT persist to the DO storage** so you will not see extra storage operations by using it. + +### Addressing + +`partyserver` exposes helper to address your DOs instead of manually through your bindings. This allows `partyserver` to implement several improvements, including a unique URL routing scheme for your DOs (e.g. `/servers/:durableClass/:durableName`). + +Compare this to the DO addressing [example above](#RPC). + +```ts +// Note the await here! +const stub = await getServerByName(env.MY_DO, "foo"); + +// We can still call RPC methods. +await stub.bar(); +``` + +Since we have a URL addressing scheme, we also get access to `routePartykitRequest()`. + +```ts + async fetch(request: Request, env: Env, ctx: ExecutionContext) { + // Behind the scenes, PartyKit normalizes your DO binding names + // and tries to do some pattern matching. + const res = await routePartykitRequest(request, env); + + if (res) return res; + + return Response("Not found", { status: 404 }); + } +``` + +You can have a look at [the implementation](https://github.com/cloudflare/partykit/blob/main/packages/partyserver/src/index.ts#L122) if you're interested. + +### `onStart` + +The extra plumbing that `Server` includes on addressing allows it to expose an `onStart` callback that is **executed every time the DO starts up** (the DO was evicted, hibernated or never created at all) and **before any `fetch` or RPC**. + +```ts +class MyServer extends Server { + onStart() { + // Some initialization logic that you wish + // to run every time the DO is started up. + const sql = this.ctx.storage.sql; + sql.exec(`...`); + } +} +``` + +### `onRequest` and `onConnect` + +`Server` already implements `fetch` for the underlying Durable Object and exposes 2 different callbacks that developers can make use of, `onRequest` and `onConnect` for HTTP requests and incoming WS connections, respectively (**WebSocket connections are accepted by default**). + +```ts +class MyServer extends Server { + async onRequest(request: Request) { + const url = new URL(request.url); + + return new Response(`Hello from ${url.origin}!`); + } + + async onConnect(conn, ctx) { + const { request } = ctx; + const url = new URL(request.url); + + // Connections are a WebSocket wrapper + conn.send(`Hello from ${url.origin}!`); + } +} +``` + +### WebSockets + +Just as `onConnect` is the callback for every new connection, `Server` also provides wrappers on top of the default callbacks from the `DurableObject` class: `onMessage`, `onClose` and `onError`. + +There's also `this.broadcast` that sends a WS message to all connected clients (no magic, just a loop over `this.getConnections()`!). + +### `this.name` + +It's hard to get a Durable Object's `name` from within it. `partyserver` tries to make it available in `this.name` but it's not a perfect solution. Read more about it [here](https://github.com/cloudflare/workerd/issues/2240). + +## Layer 2: Agent + +Now finally, the `Agent` class. `Agent` extends `Server` and provides opinionated primitives for stateful, schedulable, and observable agents that can communicate via RPC, WebSockets, and (even!) email. + +### `this.state` and `this.setState()` + +One of the core features of `Agent` is **automatic state persistence**. Developers define the shape of their state via the generic parameter and `initialState` (which is only used if no state exists in storage), and the Agent handles loading, saving, and broadcasting state changes (check `Server`'s `this.broadcast()` above). + +`this.state` is a getter that lazily loads state from storage (SQL). **State is persisted across DO evictions** when it's updated with `this.setState()`, which automatically serializes the state and writes it back to storage. +There's also `this.onStateChanged` that you can override to react to state changes. + +```ts +class MyAgent extends Agent { + initialState = { count: 0 }; + + increment() { + this.setState({ count: this.state.count + 1 }); + } + + onStateChanged(state, source) { + console.log("State updated:", state); + } +} +``` + +State is stored in the `cf_agents_state` SQL table. State messages are sent with `type: "cf_agent_state"` (both from the client and the server). Since the `agents` provides [JS and React clients](https://developers.cloudflare.com/agents/api-reference/store-and-sync-state/#synchronizing-state), real-time state updates are available out of the box. + +Protocol messages (`CF_AGENT_IDENTITY`, `CF_AGENT_STATE`, `CF_AGENT_MCP_SERVERS`) are sent automatically on connect and broadcast on changes. You can suppress these per connection by overriding `shouldSendProtocolMessages(connection, ctx)` — see [Protocol Message Control](./http-websockets.md#protocol-message-control) for details. + +### `this.sql` + +The Agent provides a convenient `sql` template tag for executing queries against the Durable Object's SQL storage. It constructs parameterized queries and executes them. This uses the **synchronous** SQL API from `this.ctx.storage.sql`. + +```ts +class MyAgent extends Agent { + onStart() { + this.sql` + CREATE TABLE IF NOT EXISTS users ( + id TEXT PRIMARY KEY, + name TEXT + ) + `; + + const userId = "1"; + const userName = "Alice"; + this.sql`INSERT INTO users (id, name) VALUES (${userId}, ${userName})`; + + const users = this.sql<{ id: string; name: string }>` + SELECT * FROM users WHERE id = ${userId} + `; + console.log(users); // [{ id: "1", name: "Alice" }] + } +} +``` + +### RPC and Callable Methods + +`agents` take Durable Objects RPC one step forward by implementing RPC through WebSockets, so clients can also call methods on the Agent directly. To make a method callable through WS, developers can use the `@callable` decorator. Methods can return a serializable value or a stream (when using `@callable({ stream: true })`). + +```ts +class MyAgent extends Agent { + @callable({ description: "Add two numbers" }) + async add(a: number, b: number) { + return a + b; + } +} +``` + +Clients can invoke this method by sending a WebSocket message: + +```json +{ + "type": "rpc", + "id": "unique-request-id", + "method": "add", + "args": [2, 3] +} +``` + +For example, with the provided `React` client it's as easy as: + +```ts +const { stub } = useAgent({ name: "my-agent" }); +const result = await stub.add(2, 3); +console.log(result); // 5 +``` + +### `this.queue` and friends + +Agents include a built-in task queue for deferred execution. This is useful for offloading work or retrying operations. The available methods are `this.queue`, `this.dequeue`, `this.dequeueAll`, `this.dequeueAllByCallback`, `this.getQueue`, and `this.getQueues`. + +```ts +class MyAgent extends Agent { + async onConnect() { + // Queue a task to be executed later + await this.queue("processTask", { userId: "123" }); + } + + async processTask(payload: { userId: string }, queueItem: QueueItem) { + console.log("Processing task for user:", payload.userId); + } +} +``` + +Tasks are stored in the `cf_agents_queues` SQL table and are automatically flushed in sequence. If a task succeeds, it's automatically dequeued. + +### `this.schedule` and friends + +Agents support scheduled execution of methods by wrapping the Durable Object's `alarm()`. The available methods are `this.schedule`, `this.getSchedule`, `this.getSchedules`, `this.cancelSchedule`. Schedules can be one-time, delayed, or recurring (using cron expressions). + +Since DOs only allow one alarm at a time, the `Agent` class works around this by managing multiple schedules in SQL and using a single alarm. + +```ts +class MyAgent extends Agent { + async foo() { + // Schedule at a specific time + await this.schedule(new Date("2025-12-25T00:00:00Z"), "sendGreeting", { + message: "Merry Christmas!" + }); + + // Schedule with a delay (in seconds) + await this.schedule(60, "checkStatus", { check: "health" }); + + // Schedule with a cron expression + await this.schedule("0 0 * * *", "dailyTask", { type: "cleanup" }); + } + + async sendGreeting(payload: { message: string }) { + console.log(payload.message); + } + + async checkStatus(payload: { check: string }) { + console.log("Running check:", payload.check); + } + + async dailyTask(payload: { type: string }) { + console.log("Daily task:", payload.type); + } +} +``` + +Schedules are stored in the `cf_agents_schedules` SQL table. Cron schedules automatically reschedule themselves after execution, while one-time schedules are deleted. + +### `this.mcp` and friends + +`Agent` includes a multi-server MCP client. This enables your Agent to interact with external services that expose MCP interfaces. The MCP client is properly documented [here](https://developers.cloudflare.com/agents/model-context-protocol/mcp-client-api/). + +```ts +class MyAgent extends Agent { + async onConnect() { + // Add an MCP server + await this.addMcpServer("GitHub", "https://mcp.example.com/sse"); + } +} +``` + +### Email Handling + +Agents can receive and reply to emails using Cloudflare's [Email Routing](https://developers.cloudflare.com/email-routing/email-workers/). + +```ts +class MyAgent extends Agent { + async onEmail(email: AgentEmail) { + console.log("Received email from:", email.from); + console.log("Subject:", email.headers.get("subject")); + + // Reply to the email + await this.replyToEmail(email, { + fromName: "My Agent", + body: "Thanks for your email!" + }); + } +} +``` + +To route emails to your Agent, use `routeAgentEmail` in your Worker's email handler: + +```ts +import { routeAgentEmail } from "agents"; +import { createAddressBasedEmailResolver } from "agents/email"; + +export default { + async email(message, env, ctx) { + await routeAgentEmail(message, env, { + resolver: createAddressBasedEmailResolver("my-agent") + }); + } +}; +``` + +For more details on email routing, resolvers, secure reply flows, and the full API, see the [Email Routing guide](./email.md). + +### Context Management + +`agents` wraps all your methods with an `AsyncLocalStorage` to maintain context throughout the request lifecycle. This allows you to access the current agent, connection, request, or email (depending of what event is being handled) from anywhere in your code: + +```ts +import { getCurrentAgent } from "agents"; + +function someUtilityFunction() { + const { agent, connection, request, email } = getCurrentAgent(); + + if (agent) { + console.log("Current agent:", agent.name); + } + + if (connection) { + console.log("WebSocket connection ID:", connection.id); + } +} +``` + +### `this.onError` + +`Agent` extends `Server`'s `onError` so it can be used to handle errors that are not necessarily WebSocket errors. It is called with a `Connection` or `unknown` error. + +```ts +class MyAgent extends Agent { + onError(connectionOrError: Connection | unknown, error?: unknown) { + if (error) { + // WebSocket connection error + console.error("Connection error:", error); + } else { + // Server error + console.error("Server error:", connectionOrError); + } + + // Optionally throw to propagate the error + throw connectionOrError; + } +} +``` + +### `this.destroy` + +`this.destroy()` drops all tables, deletes alarms, clears storage, and aborts the context. To ensure that the DO is fully evicted, `this.ctx.abort()` is called, which throws an uncatchable error that will show up in your logs (read more about it [here](https://developers.cloudflare.com/durable-objects/api/state/#abort)). + +```ts +class MyAgent extends Agent { + async onStart() { + console.log("Agent is starting up..."); + // Initialize your agent + } + + async cleanup() { + // This wipes everything! + await this.destroy(); + } +} +``` + +### `this.keepAlive` + +`this.keepAlive()` prevents the Durable Object from being evicted due to inactivity by creating a 30-second heartbeat schedule. Returns a disposer function to stop the heartbeat. For scoped work, use `this.keepAliveWhile(fn)` which automatically cleans up when the function completes. See [Keeping the Agent Alive](./scheduling.md#keeping-the-agent-alive) for full documentation. + +### Routing + +The `Agent` class re-exports PartyKit's [addressing helpers](#addressing) as `getAgentByName` and `routeAgentRequest`. + +```ts +// Same API as getServerByName +const stub = await getAgentByName(env.MY_DO, "foo"); +// ... + +// Same API as routeServerRequest +const res = await routeAgentRequest(request, env); + +if (res) return res; + +return Response("Not found", { status: 404 }); +``` diff --git a/docs/browse-the-web.md b/docs/browse-the-web.md new file mode 100644 index 0000000000..37585f3490 --- /dev/null +++ b/docs/browse-the-web.md @@ -0,0 +1,260 @@ +# Browse the Web (Experimental) + +Browser tools give your agents full access to the Chrome DevTools Protocol (CDP) through the code mode pattern. Instead of a fixed set of browser actions (click, screenshot, navigate), the LLM writes JavaScript code that runs CDP commands against a live browser session — accessing all domains, commands, events, and types in the protocol. + +Two tools are provided: + +- **`browser_search`** — query the CDP spec to discover commands, events, and types. The spec is fetched dynamically from the browser's CDP endpoint and cached for performance. +- **`browser_execute`** — run CDP commands against a live browser via a `cdp` helper. Each call opens a fresh browser session, executes the code, and closes it. + +> **Experimental** — this feature may have breaking changes in future releases. + +## When to use browser tools + +Browser tools are useful when your agent needs to: + +- **Inspect web pages** — DOM structure, computed styles, accessibility tree +- **Debug frontend issues** — network waterfalls, console errors, performance traces +- **Scrape structured data** — extract content from rendered pages +- **Capture screenshots or PDFs** — visual snapshots of web content +- **Profile performance** — Core Web Vitals, JavaScript profiling, memory analysis + +For simple page fetches where you do not need a full browser, `fetch()` is simpler. + +## Installation + +Browser tools require the Agents SDK and `@cloudflare/codemode`: + +```sh +npm install agents @cloudflare/codemode ai zod +``` + +## Quick Start + +### 1. Configure bindings + +Add the Browser Rendering and Worker Loader bindings to your `wrangler.jsonc`: + +```jsonc +// wrangler.jsonc +{ + "browser": { "binding": "BROWSER" }, + "worker_loaders": [{ "binding": "LOADER" }], + "compatibility_flags": ["nodejs_compat"] +} +``` + +### 2. Create browser tools + +```ts +import { createBrowserTools } from "agents/browser/ai"; + +const browserTools = createBrowserTools({ + browser: env.BROWSER, + loader: env.LOADER +}); +``` + +If you need to connect to a custom CDP endpoint instead of the Browser Rendering binding, pass `cdpUrl`. + +### 3. Use with streamText + +Pass browser tools alongside your other tools: + +```ts +import { streamText } from "ai"; + +const result = streamText({ + model, + system: "You are a helpful assistant that can inspect web pages.", + messages, + tools: { + ...browserTools, + ...otherTools + } +}); +``` + +When the LLM uses `browser_search`, the `code` field must be JavaScript: + +```javascript +async () => { + const s = await spec.get(); + return s.domains + .find((d) => d.name === "Network") + .commands.map((c) => ({ method: c.method, description: c.description })); +}; +``` + +When the LLM uses `browser_execute`, the `code` field must be JavaScript: + +```javascript +async () => { + const { targetId } = await cdp.send("Target.createTarget", { + url: "https://example.com" + }); + const sessionId = await cdp.attachToTarget(targetId); + const { root } = await cdp.send("DOM.getDocument", {}, { sessionId }); + const { outerHTML } = await cdp.send( + "DOM.getOuterHTML", + { + nodeId: root.nodeId + }, + { sessionId } + ); + await cdp.send("Target.closeTarget", { targetId }); + return outerHTML; +}; +``` + +## Use with an Agent + +The typical pattern is to create browser tools inside the agent's message handler: + +```ts +import { Agent } from "agents"; +import { createBrowserTools } from "agents/browser/ai"; +import { streamText, convertToModelMessages, stepCountIs } from "ai"; + +export class MyAgent extends Agent { + async onChatMessage() { + const browserTools = createBrowserTools({ + browser: this.env.BROWSER, + loader: this.env.LOADER + }); + + const result = streamText({ + model, + system: "You can browse the web and inspect pages.", + messages: await convertToModelMessages(this.messages), + tools: { + ...browserTools, + ...this.mcp.getAITools() + }, + stopWhen: stepCountIs(10) + }); + + return result.toUIMessageStreamResponse(); + } +} +``` + +## TanStack AI + +For TanStack AI, use the `/tanstack-ai` export: + +```ts +import { createBrowserTools } from "agents/browser/tanstack-ai"; +import { chat } from "@tanstack/ai"; + +const browserTools = createBrowserTools({ + browser: env.BROWSER, + loader: env.LOADER +}); + +const stream = chat({ + adapter: openaiText("gpt-4o"), + tools: [...browserTools, ...otherTools], + messages +}); +``` + +## Execution model + +- `browser_search` fetches the live CDP protocol from the browser's `/json/protocol` endpoint and caches it briefly. +- `browser_execute` opens a fresh browser session for the call, exposes a small `cdp` helper API to sandboxed code, and closes the session when execution finishes. +- LLM-generated code runs in a Worker sandbox. CDP traffic stays in the host worker. + +## CDP helper API + +Inside `browser_execute`, the following functions are available: + +### `cdp.send(method, params?, options?)` + +Send a CDP command and wait for the response. + +| Parameter | Type | Description | +| ------------------- | --------- | --------------------------------------------------------- | +| `method` | `string` | CDP method (e.g. `"DOM.getDocument"`, `"Network.enable"`) | +| `params` | `unknown` | Method parameters | +| `options.timeoutMs` | `number` | Per-command timeout (default: 10s) | +| `options.sessionId` | `string` | Target session ID (required for page-scoped commands) | + +### `cdp.attachToTarget(targetId, options?)` + +Attach to a target and get a session ID. Uses `Target.attachToTarget` with `flatten: true`. + +| Parameter | Type | Description | +| ------------------- | -------- | ------------------------------ | +| `targetId` | `string` | The target to attach to | +| `options.timeoutMs` | `number` | Timeout for the attach command | + +Returns the `sessionId` string. + +### `cdp.getDebugLog(limit?)` + +Get recent CDP debug log entries (sends, receives, errors). Defaults to the last 50 entries, max 400. + +### `cdp.clearDebugLog()` + +Clear the debug log buffer. + +## Configuration + +### `createBrowserTools(options)` + +Returns AI SDK tools (`browser_search` and `browser_execute`). + +| Option | Type | Default | Description | +| ------------ | ------------------------ | -------- | ------------------------------------------------------ | +| `browser` | `Fetcher` | — | Browser Rendering binding | +| `cdpUrl` | `string` | — | Optional override for a custom CDP endpoint | +| `cdpHeaders` | `Record` | — | Headers for CDP URL discovery (e.g. Cloudflare Access) | +| `loader` | `WorkerLoader` | required | Worker Loader binding for sandboxed execution | +| `timeout` | `number` | `30000` | Execution timeout in milliseconds | + +Either `browser` or `cdpUrl` must be provided. When both are set, `cdpUrl` takes priority. + +### Raw access + +For custom integrations, import the building blocks directly: + +```ts +import { + CdpSession, + connectBrowser, + connectUrl, + createBrowserToolHandlers +} from "agents/browser"; + +// Connect to a custom CDP endpoint +const session = await connectUrl("http://localhost:9222"); +const version = await session.send("Browser.getVersion"); +session.close(); +``` + +## Local development + +Recent Wrangler releases support Browser Rendering in local development. `npx wrangler dev` provisions the browser automatically, so the same `browser: env.BROWSER` setup works locally and when deployed. + +Use `cdpUrl` only when you intentionally want to connect to some other CDP-compatible browser endpoint, such as a tunnel or a manually managed Chrome instance. + +## Security considerations + +- LLM-generated code runs in **isolated Worker sandboxes** — each execution gets its own Worker instance +- External network access (`fetch`, `connect`) is **blocked** in the sandbox at the runtime level +- CDP commands are dispatched via Workers RPC — the WebSocket lives in the host, not the sandbox +- The CDP spec stays on the server — only query results flow to the LLM +- Responses are truncated to approximately 6,000 tokens to prevent context window overflow + +## Current limitations + +- **One session per execute call** — each `browser_execute` invocation opens a fresh browser session. Multi-step workflows must be completed within a single code block. +- **Local development depends on Wrangler support** — if Browser Rendering local mode is unavailable in your environment, upgrade Wrangler or provide `cdpUrl` explicitly. +- **No authenticated sessions** — the browser starts without any cookies or login state. A future Browser Isolation integration could enable user-authenticated sessions. +- Requires `@cloudflare/codemode` as a peer dependency +- Limited to JavaScript execution in the sandbox (no TypeScript syntax) + +## Example + +See [`examples/ai-chat/`](../examples/ai-chat/) for a working example that combines browser tools with other AI SDK tools, MCP servers, and tool approval. diff --git a/docs/callable-methods.md b/docs/callable-methods.md new file mode 100644 index 0000000000..a49852f041 --- /dev/null +++ b/docs/callable-methods.md @@ -0,0 +1,627 @@ +# Callable Methods + +Callable methods let clients invoke agent methods over WebSocket using RPC (Remote Procedure Call). Mark methods with `@callable()` to expose them to external clients like browsers, mobile apps, or other services. + +## Overview + +```typescript +import { Agent, callable } from "agents"; + +export class MyAgent extends Agent { + @callable() + async greet(name: string): Promise { + return `Hello, ${name}!`; + } +} +``` + +```typescript +// Client +const result = await agent.stub.greet("World"); +console.log(result); // "Hello, World!" +``` + +### How It Works + +``` +┌─────────┐ ┌─────────┐ +│ Client │ │ Agent │ +└────┬────┘ └────┬────┘ + │ │ + │ agent.stub.greet("World") │ + │ ──────────────────────────────────▶ │ + │ WebSocket RPC message │ + │ │ + │ Check @callable + │ Execute method + │ │ + │ ◀──────────────────────────────── │ + │ "Hello, World!" │ + │ │ +``` + +### When to Use @callable + +| Scenario | Use | +| ------------------------------------ | ----------------------------- | +| Browser/mobile calling agent | `@callable()` | +| External service calling agent | `@callable()` | +| Worker calling agent (same codebase) | DO RPC (no decorator needed) | +| Agent calling another agent | DO RPC via `getAgentByName()` | + +The `@callable()` decorator is specifically for WebSocket-based RPC from external clients. When calling from within the same Worker or another agent, use standard [Durable Object RPC](https://developers.cloudflare.com/durable-objects/best-practices/create-durable-object-stubs-and-send-requests/) directly. + +## TypeScript and Vite Configuration + +The `@callable()` decorator uses TC39 standard decorators, which require two build-time configurations: + +**1. Add the `agents/vite` plugin** — Vite 8 uses Oxc for transpilation, which does not yet support TC39 decorators ([oxc#9170](https://github.com/oxc-project/oxc/issues/9170)). The plugin adds the required Babel transform: + +```typescript +// vite.config.ts +import agents from "agents/vite"; + +export default defineConfig({ + plugins: [agents(), react(), cloudflare()] +}); +``` + +The plugin only runs the transform on files containing `@` syntax. It is safe to include even if you do not use decorators. + +**2. Extend `agents/tsconfig`** — this sets `target: "ES2021"` and all other recommended compiler options: + +```json +{ + "extends": "agents/tsconfig" +} +``` + +If you cannot extend the shared config, set `"target": "ES2021"` manually in your `tsconfig.json`. + +Without both of these, your dev server will fail with `SyntaxError: Invalid or unexpected token`. + +> **Warning:** Do not set `"experimentalDecorators": true` in your `tsconfig.json`. The Agents SDK uses [TC39 standard decorators](https://github.com/tc39/proposal-decorators), not TypeScript legacy decorators. Enabling `experimentalDecorators` applies an incompatible transform that silently breaks `@callable()` at runtime. + +## Basic Usage + +### Defining Callable Methods + +Add the `@callable()` decorator to any method you want to expose: + +```typescript +import { Agent, callable } from "agents"; + +type State = { + count: number; + items: string[]; +}; + +export class CounterAgent extends Agent { + initialState: State = { count: 0, items: [] }; + + @callable() + increment(): number { + this.setState({ ...this.state, count: this.state.count + 1 }); + return this.state.count; + } + + @callable() + decrement(): number { + this.setState({ ...this.state, count: this.state.count - 1 }); + return this.state.count; + } + + @callable() + async addItem(item: string): Promise { + this.setState({ ...this.state, items: [...this.state.items, item] }); + return this.state.items; + } + + @callable() + getStats(): { count: number; itemCount: number } { + return { + count: this.state.count, + itemCount: this.state.items.length + }; + } +} +``` + +### Calling from the Client + +There are two ways to call methods from the client: + +**Using `agent.stub` (recommended):** + +```typescript +// Clean, typed syntax +const count = await agent.stub.increment(); +const items = await agent.stub.addItem("new item"); +const stats = await agent.stub.getStats(); +``` + +**Using `agent.call()`:** + +```typescript +// Explicit method name as string +const count = await agent.call("increment"); +const items = await agent.call("addItem", ["new item"]); +const stats = await agent.call("getStats"); +``` + +The `stub` proxy provides better ergonomics and TypeScript support. + +## Method Signatures + +### Serializable Types + +Arguments and return values must be JSON-serializable: + +```typescript +// ✅ Valid - primitives and plain objects +@callable() +processData(input: { name: string; count: number }): { result: boolean } { + return { result: true }; +} + +// ✅ Valid - arrays +@callable() +processItems(items: string[]): number[] { + return items.map(item => item.length); +} + +// ❌ Invalid - non-serializable types +@callable() +badMethod(fn: Function, date: Date): Map { + // Functions, Dates, Maps, Sets, etc. cannot be serialized +} +``` + +### Async Methods + +Both sync and async methods work: + +```typescript +// Sync method +@callable() +add(a: number, b: number): number { + return a + b; +} + +// Async method +@callable() +async fetchUser(id: string): Promise { + const user = await this.sql`SELECT * FROM users WHERE id = ${id}`; + return user[0]; +} +``` + +### Void Methods + +Methods that don't return a value: + +```typescript +@callable() +async logEvent(event: string): Promise { + await this.sql`INSERT INTO events (name) VALUES (${event})`; +} +``` + +On the client, these still return a Promise that resolves when the method completes: + +```typescript +await agent.stub.logEvent("user-clicked"); +// Resolves when the server confirms execution +``` + +## Streaming Responses + +For methods that produce data over time (like AI text generation), use streaming: + +### Defining a Streaming Method + +```typescript +import { Agent, callable, type StreamingResponse } from "agents"; + +export class AIAgent extends Agent { + @callable({ streaming: true }) + async generateText(stream: StreamingResponse, prompt: string) { + // First parameter is always StreamingResponse for streaming methods + + for await (const chunk of this.llm.stream(prompt)) { + stream.send(chunk); // Send each chunk to the client + } + + stream.end(); // Signal completion + } + + @callable({ streaming: true }) + async streamNumbers(stream: StreamingResponse, count: number) { + for (let i = 0; i < count; i++) { + stream.send(i); + await new Promise((resolve) => setTimeout(resolve, 100)); + } + stream.end(count); // Optional final value + } +} +``` + +### Consuming Streams on the Client + +```typescript +// Preferred format (supports timeout and other options) +await agent.call("generateText", [prompt], { + stream: { + onChunk: (chunk) => { + // Called for each chunk + appendToOutput(chunk); + }, + onDone: (finalValue) => { + // Called when stream ends + console.log("Stream complete", finalValue); + }, + onError: (error) => { + // Called if an error occurs + console.error("Stream error:", error); + } + } +}); + +// Legacy format (still supported for backward compatibility) +await agent.call("generateText", [prompt], { + onChunk: (chunk) => appendToOutput(chunk), + onDone: (finalValue) => console.log("Done", finalValue), + onError: (error) => console.error("Error:", error) +}); +``` + +### StreamingResponse API + +| Method | Description | +| ------------------ | ------------------------------------------------ | +| `send(chunk)` | Send a chunk to the client | +| `end(finalChunk?)` | End the stream, optionally with a final value | +| `error(message)` | Send an error to the client and close the stream | + +```typescript +@callable({ streaming: true }) +async processWithProgress(stream: StreamingResponse, items: string[]) { + for (let i = 0; i < items.length; i++) { + await this.process(items[i]); + stream.send({ progress: (i + 1) / items.length, item: items[i] }); + } + stream.end({ completed: true, total: items.length }); +} +``` + +## TypeScript Integration + +### Typed Client Calls + +Pass your agent class as a type parameter for full type safety: + +```typescript +import { useAgent } from "agents/react"; +import type { MyAgent } from "./server"; + +function App() { + const agent = useAgent({ + agent: "MyAgent", + name: "default" + }); + + // ✅ TypeScript knows the method signature + const result = await agent.stub.greet("World"); + // ^? string + + // ✅ TypeScript catches errors + await agent.stub.greet(123); // Error: Argument of type 'number' is not assignable + await agent.stub.nonExistent(); // Error: Property 'nonExistent' does not exist +} +``` + +### Excluding Non-Callable Methods + +If you have methods that aren't decorated with `@callable()`, you can exclude them from the type: + +```typescript +class MyAgent extends Agent { + @callable() + publicMethod(): string { + return "public"; + } + + // Not callable from clients + internalMethod(): void { + // internal logic + } +} + +// Exclude internal methods from the client type +const agent = useAgent, {}>({ + agent: "MyAgent" +}); + +agent.stub.publicMethod(); // ✅ Works +agent.stub.internalMethod(); // ✅ TypeScript error +``` + +### Type Inference for State + +When methods return `this.state`, TypeScript correctly infers the type: + +```typescript +type MyState = { count: number; name: string }; + +class MyAgent extends Agent { + @callable() + async getState(): Promise { + return this.state; + } +} + +// Client +const state = await agent.stub.getState(); +// ^? MyState +``` + +## Error Handling + +### Throwing Errors in Callable Methods + +Errors thrown in callable methods are propagated to the client: + +```typescript +@callable() +async riskyOperation(data: unknown): Promise { + if (!isValid(data)) { + throw new Error("Invalid data format"); + } + + try { + await this.processData(data); + } catch (e) { + throw new Error("Processing failed: " + e.message); + } +} +``` + +### Client-Side Error Handling + +```typescript +try { + const result = await agent.stub.riskyOperation(data); +} catch (error) { + // Error thrown by the agent method + console.error("RPC failed:", error.message); +} +``` + +### Streaming Error Handling + +For streaming methods, use the `onError` callback: + +```typescript +await agent.call("streamData", [input], { + stream: { + onChunk: (chunk) => handleChunk(chunk), + onError: (errorMessage) => { + console.error("Stream error:", errorMessage); + showErrorUI(errorMessage); + }, + onDone: (result) => handleComplete(result) + } +}); +``` + +Server-side, you can use `stream.error()` to gracefully send an error mid-stream: + +```typescript +@callable({ streaming: true }) +async processItems(stream: StreamingResponse, items: string[]) { + for (const item of items) { + try { + const result = await this.process(item); + stream.send(result); + } catch (e) { + stream.error(`Failed to process ${item}: ${e.message}`); + return; // Stream is now closed + } + } + stream.end(); +} +``` + +### Connection Errors + +If the WebSocket connection closes while RPC calls are pending, they automatically reject with a "Connection closed" error: + +```typescript +try { + const result = await agent.call("longRunningMethod", []); +} catch (error) { + if (error.message === "Connection closed") { + // Handle disconnection + console.log("Lost connection to agent"); + } +} +``` + +#### Retrying After Reconnection + +PartySocket automatically reconnects after disconnection. To retry a failed call after reconnection, await `agent.ready` before retrying: + +```typescript +async function callWithRetry( + agent: AgentClient, + method: string, + args: unknown[] = [] +): Promise { + try { + return await agent.call(method, args); + } catch (error) { + if (error.message === "Connection closed") { + await agent.ready; // Wait for reconnection + return await agent.call(method, args); // Retry once + } + throw error; + } +} + +// Usage +const result = await callWithRetry(agent, "processData", [data]); +``` + +> **Note:** Only retry idempotent operations. If the server received the request but the connection dropped before the response arrived, retrying could cause duplicate execution. + +## When NOT to Use @callable + +### Worker-to-Agent Calls + +When calling an agent from the same Worker (e.g., in your `fetch` handler), use Durable Object RPC directly: + +```typescript +import { getAgentByName } from "agents"; + +export default { + async fetch(request: Request, env: Env) { + // Get the agent stub + const agent = await getAgentByName(env.MyAgent, "instance-name"); + + // Call methods directly - no @callable needed + const result = await agent.processData(data); + + return Response.json(result); + } +}; +``` + +### Agent-to-Agent Calls + +When one agent needs to call another: + +```typescript +class OrchestratorAgent extends Agent { + async delegateWork(taskId: string) { + // Get another agent + const worker = await getAgentByName(this.env.WorkerAgent, taskId); + + // Call its methods directly + const result = await worker.doWork(); + + return result; + } +} +``` + +### Why the Distinction? + +| RPC Type | Transport | Use Case | +| ----------- | --------- | --------------------------------- | +| `@callable` | WebSocket | External clients (browsers, apps) | +| DO RPC | Internal | Worker ↔ Agent, Agent ↔ Agent | + +DO RPC is more efficient for internal calls since it doesn't go through WebSocket serialization. The `@callable` decorator adds the necessary WebSocket RPC handling for external clients. + +## API Reference + +### `@callable(metadata?)` Decorator + +Marks a method as callable from external clients. + +```typescript +import { callable } from "agents"; + +@callable() +method(): void {} + +@callable({ streaming: true }) +streamingMethod(stream: StreamingResponse): void {} + +@callable({ description: "Fetches user data" }) +getUser(id: string): User {} +``` + +### `CallableMetadata` Type + +```typescript +type CallableMetadata = { + /** Optional description of what the method does */ + description?: string; + /** Whether the method supports streaming responses */ + streaming?: boolean; +}; +``` + +### `StreamingResponse` Class + +Used in streaming callable methods to send data to the client. + +```typescript +import { type StreamingResponse } from "agents"; + +@callable({ streaming: true }) +async streamData(stream: StreamingResponse, input: string) { + stream.send("chunk 1"); + stream.send("chunk 2"); + stream.end("final"); +} +``` + +| Method | Signature | Description | +| ------- | -------------------------------- | ---------------------------------- | +| `send` | `(chunk: unknown) => void` | Send a chunk to the client | +| `end` | `(finalChunk?: unknown) => void` | End the stream | +| `error` | `(message: string) => void` | Send an error and close the stream | + +### Client Methods + +| Method | Signature | Description | +| ------------ | -------------------------------------- | --------------------- | +| `agent.call` | `(method, args?, options?) => Promise` | Call a method by name | +| `agent.stub` | `Proxy` | Typed method calls | + +```typescript +// Using call() +await agent.call("methodName", [arg1, arg2]); +await agent.call("streamMethod", [arg], { + stream: { onChunk, onDone, onError } +}); + +// With timeout (rejects if call doesn't complete in time) +await agent.call("slowMethod", [], { timeout: 5000 }); + +// Using stub +await agent.stub.methodName(arg1, arg2); +``` + +### `CallOptions` Type + +```typescript +type CallOptions = { + /** Timeout in milliseconds. Rejects if call doesn't complete in time. */ + timeout?: number; + /** Streaming options */ + stream?: { + onChunk?: (chunk: unknown) => void; + onDone?: (finalChunk: unknown) => void; + onError?: (error: string) => void; + }; +}; +``` + +> **Backward Compatibility**: The legacy format `{ onChunk, onDone, onError }` (without nesting under `stream`) is still supported. The client auto-detects which format you're using. + +### `getCallableMethods()` Method + +Returns a map of all callable methods on the agent with their metadata. Useful for introspection and auto-documentation. + +```typescript +const methods = agent.getCallableMethods(); +// Map + +for (const [name, meta] of methods) { + console.log(`${name}: ${meta.description || "(no description)"}`); + if (meta.streaming) console.log(" (streaming)"); +} +``` diff --git a/docs/chat-agents.md b/docs/chat-agents.md new file mode 100644 index 0000000000..cf952dea12 --- /dev/null +++ b/docs/chat-agents.md @@ -0,0 +1,1495 @@ +# Chat Agents + +Build AI-powered chat interfaces with `AIChatAgent` and `useAgentChat`. Messages are automatically persisted to SQLite, streams resume on disconnect, and tool calls work across server and client. + +## Overview + +`@cloudflare/ai-chat` provides two main exports: + +| Export | Import | Purpose | +| -------------- | --------------------------- | -------------------------------------------------------------- | +| `AIChatAgent` | `@cloudflare/ai-chat` | Server-side agent class with message persistence and streaming | +| `useAgentChat` | `@cloudflare/ai-chat/react` | React hook for building chat UIs | + +Built on the [AI SDK](https://ai-sdk.dev) and Cloudflare Durable Objects, you get: + +- **Automatic message persistence** — conversations stored in SQLite, survive restarts +- **Resumable streaming** — disconnected clients resume mid-stream without data loss +- **Real-time sync** — messages broadcast to all connected clients via WebSocket +- **Tool support** — server-side, client-side, and human-in-the-loop tool patterns +- **Data parts** — attach typed JSON (citations, progress, usage) to messages alongside text +- **Row size protection** — automatic compaction when messages approach SQLite limits + +## Quick Start + +### Install + +```sh +npm install @cloudflare/ai-chat agents ai workers-ai-provider +``` + +### Server + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { createWorkersAI } from "workers-ai-provider"; +import { streamText, convertToModelMessages } from "ai"; + +export class ChatAgent extends AIChatAgent { + async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages) + }); + + return result.toUIMessageStreamResponse(); + } +} +``` + +### Client + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "ChatAgent" }); + const { messages, sendMessage, status } = useAgentChat({ agent }); + + return ( +
+ {messages.map((msg) => ( +
+ {msg.role}: + {msg.parts.map((part, i) => + part.type === "text" ? {part.text} : null + )} +
+ ))} + +
{ + e.preventDefault(); + const input = e.currentTarget.elements.namedItem( + "input" + ) as HTMLInputElement; + sendMessage({ text: input.value }); + input.value = ""; + }} + > + + +
+
+ ); +} +``` + +### Wrangler Config + +```jsonc +// wrangler.jsonc +{ + "ai": { "binding": "AI" }, + "durable_objects": { + "bindings": [{ "name": "ChatAgent", "class_name": "ChatAgent" }] + }, + "migrations": [{ "tag": "v1", "new_sqlite_classes": ["ChatAgent"] }] +} +``` + +The `new_sqlite_classes` migration is required — `AIChatAgent` uses SQLite for message persistence and stream chunk buffering. + +## How It Works + +``` +┌──────────┐ WebSocket ┌──────────────┐ +│ Client │ ◀──────────────────────────────────▶ │ AIChatAgent │ +│ │ │ │ +│ useAgent │ CF_AGENT_USE_CHAT_REQUEST ──────▶ │ onChatMessage│ +│ Chat │ │ │ +│ │ ◀────── CF_AGENT_USE_CHAT_RESPONSE │ streamText │ +│ │ (UIMessageChunk stream) │ │ +│ │ │ SQLite │ +│ │ ◀────── CF_AGENT_CHAT_MESSAGES │ (messages, │ +│ │ (broadcast to all clients) │ chunks) │ +└──────────┘ └──────────────┘ +``` + +1. The client sends a message via WebSocket +2. `AIChatAgent` persists messages to SQLite and calls your `onChatMessage` method +3. Your method returns a streaming `Response` (typically from `streamText`) +4. Chunks stream back over WebSocket in real-time +5. When the stream completes, the final message is persisted and broadcast to all connections + +## Server API + +### `AIChatAgent` + +Extends `Agent` from the `agents` package. Manages conversation state, persistence, and streaming. + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; + +export class ChatAgent extends AIChatAgent { + // Access current messages + // this.messages: UIMessage[] + + // Limit stored messages (optional) + maxPersistedMessages = 200; + + // Collapse overlapping user submits to the latest one (optional) + // messageConcurrency = "latest"; + + async onChatMessage(onFinish?, options?) { + // onFinish: optional callback for streamText (cleanup is automatic) + // options.abortSignal: cancel signal + // options.body: custom data from client + // Return a Response (streaming or plain text) + } +} +``` + +### `onChatMessage` + +This is the main method you override. It receives the conversation context and should return a `Response`. + +**Streaming response** (most common): + +```typescript +async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + system: "You are a helpful assistant.", + messages: await convertToModelMessages(this.messages) + }); + + return result.toUIMessageStreamResponse(); +} +``` + +**Plain text response**: + +```typescript +async onChatMessage() { + return new Response("Hello! I am a simple agent.", { + headers: { "Content-Type": "text/plain" } + }); +} +``` + +**Accessing custom body data and request ID**: + +```typescript +async onChatMessage(_onFinish, options) { + const { timezone, userId } = options?.body ?? {}; + // Use these values in your LLM call or business logic + + // options.requestId — unique identifier for this chat request, + // useful for logging and correlating events + console.log("Request ID:", options?.requestId); +} +``` + +### `this.messages` + +The current conversation history, loaded from SQLite. This is an array of `UIMessage` objects from the AI SDK. Messages are automatically persisted after each interaction. + +### `maxPersistedMessages` + +Cap the number of messages stored in SQLite. When the limit is exceeded, the oldest messages are deleted. This controls storage only — it does not affect what is sent to the LLM. + +```typescript +export class ChatAgent extends AIChatAgent { + maxPersistedMessages = 200; +} +``` + +To control what is sent to the model, use the AI SDK's `pruneMessages()`: + +```typescript +import { streamText, convertToModelMessages, pruneMessages } from "ai"; + +async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: pruneMessages({ + messages: await convertToModelMessages(this.messages), + reasoning: "before-last-message", + toolCalls: "before-last-2-messages" + }) + }); + + return result.toUIMessageStreamResponse(); +} +``` + +### `messageConcurrency` + +Controls what happens when a new `sendMessage()` submit arrives while another +chat turn is already active or queued. + +| Value | Behavior | +| ------------------------------- | -------------------------------------------------------------------- | +| `"queue"` | Process every submit in order (default, existing behavior) | +| `"latest"` | Keep only the newest overlapping submit | +| `"merge"` | Collapse overlapping queued user messages into one follow-up submit | +| `"drop"` | Ignore overlapping submits entirely | +| `{ strategy: "debounce", ... }` | Wait for a quiet period, then run only the latest overlapping submit | + +```typescript +export class ChatAgent extends AIChatAgent { + messageConcurrency = "latest"; +} +``` + +Debounce uses the same trailing-edge semantics as chat apps that wait for the +user to finish sending a burst of short messages: + +```typescript +export class ChatAgent extends AIChatAgent { + messageConcurrency = { + strategy: "debounce", + debounceMs: 1000 + }; +} +``` + +**Choosing a strategy:** + +- Building a focused assistant where each turn matters? Use `"latest"` — the user can correct themselves mid-stream and only the final message gets a response. +- Building a messaging or chat app where every message should be processed? Use `"queue"` (default) or `"merge"` to collapse rapid-fire messages into one turn. +- Want to prevent accidental double-sends? Use `"drop"` — overlapping submits are rejected and the client rolls back. +- Users send bursts of short messages (like a messaging app)? Use `"debounce"` to wait for a quiet window before responding. + +**What the user sees:** + +- `"queue"` — every message gets its own assistant response, in order. Standard chat behavior. +- `"latest"` — all user messages appear in the transcript, but only the last overlapping message gets an assistant response. Earlier overlapping messages sit in the history with no reply. +- `"merge"` — overlapping user messages are collapsed into one combined message in the transcript, which gets a single assistant response. +- `"drop"` — the overlapping message briefly appears (optimistic), then disappears when the server sends back the rollback. +- `"debounce"` — same as `"latest"`, but the response waits for a quiet period before starting. + +Notes: + +- This setting only applies to overlapping `sendMessage()` submits (`trigger: "submit-message"`) +- `regenerate()`, tool continuations, approvals, clears, and programmatic `saveMessages()` calls keep the existing serialized behavior +- `"latest"` and `"debounce"` still persist the skipped user messages in `this.messages`; they only suppress extra model turns +- `"drop"` rejects the overlapping submit before it is persisted + +### `waitForMcpConnections` + +Controls whether `AIChatAgent` waits for MCP server connections to settle before calling `onChatMessage`. This ensures `this.mcp.getAITools()` returns the full set of tools, especially after Durable Object hibernation when connections are being restored in the background. + +| Value | Behavior | +| --------------------- | --------------------------------------------- | +| `{ timeout: 10_000 }` | Wait up to 10 seconds (default) | +| `{ timeout: N }` | Wait up to `N` milliseconds | +| `true` | Wait indefinitely until all connections ready | +| `false` | Do not wait (old behavior before 0.2.0) | + +```typescript +export class ChatAgent extends AIChatAgent { + // Default — waits up to 10 seconds + // waitForMcpConnections = { timeout: 10_000 }; + + // Wait forever + waitForMcpConnections = true; + + // Disable waiting + waitForMcpConnections = false; +} +``` + +For lower-level control, call `this.mcp.waitForConnections()` directly inside your `onChatMessage` instead. + +### `persistMessages` and `saveMessages` + +For advanced cases (schedule callbacks, webhook handlers, background tasks), +you can manually persist messages or queue programmatic turns: + +```typescript +// Persist messages without triggering a new response +await this.persistMessages(messages); + +// Persist messages AND trigger onChatMessage +await this.saveMessages([...this.messages, newMessage]); + +// Functional form: derive from latest transcript at execution time +// (e.g., from a schedule() callback or webhook handler) +await this.saveMessages((messages) => [...messages, syntheticMessage]); +``` + +Use the functional form when background work needs to append or transform +messages against the latest persisted transcript when the turn actually starts. +This avoids stale baselines when multiple `saveMessages()` calls queue up +behind active work. + +`saveMessages()` returns `{ requestId, status }` so callers can detect whether +the turn ran (`"completed"`) or was skipped because the chat was cleared +(`"skipped"`). + +### `onChatResponse` + +Called after a chat turn completes and the assistant message has been persisted. Override this to react when the agent finishes responding — broadcast state, process queued work, track analytics, or trigger follow-up messages. + +```typescript +import { AIChatAgent, type ChatResponseResult } from "@cloudflare/ai-chat"; + +export class ChatAgent extends AIChatAgent { + protected async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed") { + this.broadcast(JSON.stringify({ streaming: false })); + } + } +} +``` + +The turn lock is released before `onChatResponse` runs, so it is safe to call `saveMessages` from inside the hook. This enables sequential queue processing: + +```typescript +protected async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed" && this.workQueue.length > 0) { + const next = this.workQueue.shift()!; + await this.saveMessages([ + ...this.messages, + { id: nanoid(), role: "user", parts: [{ type: "text", text: next }] } + ]); + } +} +``` + +When `saveMessages` is called from `onChatResponse`, the inner turn's response is automatically drained — `onChatResponse` fires again for the inner response, allowing the queue to progress naturally. This continues until the queue is empty. + +Responses triggered from inside `onChatResponse` do not fire the hook concurrently. They are drained sequentially after the outer hook returns. + +**`ChatResponseResult` fields:** + +| Field | Type | Description | +| -------------- | ------------------------------------- | ---------------------------------------------------- | +| `message` | `UIMessage` | The finalized assistant message from this turn | +| `requestId` | `string` | The request ID associated with this turn | +| `continuation` | `boolean` | Whether this turn was a continuation (auto-continue) | +| `status` | `"completed" \| "error" \| "aborted"` | How the turn ended | +| `error` | `string \| undefined` | Error message when `status` is `"error"` | + +`onChatResponse` fires for all turn completion paths: WebSocket chat requests, `saveMessages`, and auto-continuation after tool results or approvals. + +### Turn coordination helpers + +`AIChatAgent` serializes chat turns — WebSocket requests, tool continuations, +`saveMessages()` calls all run one at a time. Most subclasses do not need to +think about this; the SDK handles the queuing automatically. If you configure +`messageConcurrency`, that policy decides which overlapping `sendMessage()` submits +make it into the queue. + +Coordination becomes relevant when your subclass runs code **outside** the +normal `onChatMessage()` flow — for example, a `schedule()` callback that +injects a message, a workflow-switching method, or a custom clear handler that +scopes deletes to a particular workflow. In these situations, the subclass +needs to know whether the conversation is mid-stream or waiting on user input +before it can safely act on `this.messages`. + +Three protected helpers cover these cases: + +| Helper | Returns | When to use | +| ------------------------- | ------------------ | ------------------------------------------------------------------------------------------------------------------------------- | +| `waitUntilStable()` | `Promise` | Before reading or writing messages — waits for any active stream, pending tool interactions, and queued continuations to finish | +| `resetTurnState()` | `void` | When discarding the current conversation state — aborts the active stream and invalidates queued continuations | +| `hasPendingInteraction()` | `boolean` | When you need a synchronous check for whether a tool is waiting on user input or approval | + +#### How the turn lifecycle works + +A chat turn starts when a WebSocket message or `saveMessages()` call enters the +queue and ends after the `_reply()` stream +finishes and the final assistant message is persisted. If a tool result or +approval arrives with +`autoContinue: true`, a continuation turn is queued automatically — the +conversation is not stable until that continuation finishes too. + +A pending interaction is different from an active turn. The stream has finished, +but a tool part in the assistant message is in `input-available` or +`approval-requested` state — the SDK is waiting for the client to send a result +or approval. Until the user responds, `this.messages` reflects the pending +state. + +`waitUntilStable()` handles both cases: it drains the turn queue, checks for +pending interactions, waits for any in-flight tool applies, and loops until +nothing is left. It only returns `true` once the conversation is genuinely +idle. + +#### Waiting before injecting messages + +The most common pattern is a `schedule()` callback or `onConnect()` method that +needs to inject a synthetic message into the conversation. Call +`waitUntilStable()` before reading `this.messages` or calling `saveMessages()`: + +```typescript +async onTaskComplete(payload: { result: string }) { + const ready = await this.waitUntilStable({ timeout: 30_000 }); + if (!ready) return; // timed out — a pending interaction was not resolved + + const syntheticMessage = { + id: nanoid(), + role: "user" as const, + parts: [{ type: "text" as const, text: `Task result: ${payload.result}` }] + }; + + await this.saveMessages((messages) => [...messages, syntheticMessage]); +} +``` + +Always pass a `timeout`. Without one, `waitUntilStable()` waits indefinitely — +if a tool is pending user approval and the user closes their browser, the +promise never resolves. The timeout lets you fail gracefully and retry later. + +When `waitUntilStable()` returns `true`, `this.messages` is safe to read and +`saveMessages()` will not overlap with another turn. When it returns `false`, +the conversation is still in flux and you should not assume message state is +settled. + +#### Checking pending interactions synchronously + +`hasPendingInteraction()` is a synchronous check — it scans `this.messages` +for any assistant message with a tool part in `input-available` or +`approval-requested` state. Use it when you need to branch without awaiting: + +```typescript +async onConnect(connection, ctx) { + if (this.hasPendingInteraction()) { + connection.send(JSON.stringify({ + type: "status", + message: "Waiting for your input on a pending tool action" + })); + } +} +``` + +This does not tell you whether a stream is active — only whether a tool is +waiting on user input. For most coordination needs, prefer `waitUntilStable()`. + +#### Resetting on workflow switch + +Call `resetTurnState()` when the user switches context and the current stream +and any queued continuations are no longer relevant. It does three things: +increments the internal epoch (so queued continuations skip themselves), fires +the abort signal on the active stream, and clears any pending interaction +bookkeeping. + +```typescript +async switchWorkflow(newWorkflowId: string) { + this.resetTurnState(); + this.workflowId = newWorkflowId; +} +``` + +After `resetTurnState()`, the turn queue drains quickly — aborted turns finish +their cleanup and skipped continuations return immediately. + +#### Overriding the clear handler + +The SDK's built-in `CF_AGENT_CHAT_CLEAR` handler calls `resetTurnState()` +automatically. If your `onMessage` override intercepts `CF_AGENT_CHAT_CLEAR` +and returns before the SDK sees the message — for example, to scope the delete +to a specific workflow — the built-in handler never runs. The active stream +continues and queued continuations persist into the newly-cleared conversation. + +Call `this.resetTurnState()` before performing your scoped delete: + +```typescript +import { MessageType } from "@cloudflare/ai-chat/types"; + +const _onMessage = this.onMessage.bind(this); +this.onMessage = async (connection, message) => { + if (typeof message === "string") { + const data = JSON.parse(message); + if (data.type === MessageType.CF_AGENT_CHAT_CLEAR) { + this.resetTurnState(); + this.sql` + DELETE FROM cf_ai_chat_agent_messages + WHERE workflow_id = ${this.workflowId} + `; + this.messages = []; + return; + } + } + return _onMessage(connection, message); +}; +``` + +### Lifecycle Hooks + +Override `onConnect` and `onClose` to add custom logic. Stream resumption and message sync are handled for you automatically — you do not need to call `super`: + +```typescript +export class ChatAgent extends AIChatAgent { + async onConnect(connection, ctx) { + // Your custom logic (e.g., logging, auth checks) + console.log("Client connected:", connection.id); + // Stream resumption and message sync are handled automatically + } + + async onClose(connection, code, reason, wasClean) { + console.log("Client disconnected:", connection.id); + // Connection cleanup is handled automatically + } +} +``` + +The `destroy()` method cancels any pending chat requests and cleans up stream state. It is called automatically when the Durable Object is evicted, but you can call it manually if needed. + +### Request Cancellation + +When a user clicks "stop" in the chat UI, the client sends a `CF_AGENT_CHAT_REQUEST_CANCEL` message. The server propagates this to the `abortSignal` in `options`: + +```typescript +async onChatMessage(_onFinish, options) { + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages), + abortSignal: options?.abortSignal // Pass through for cancellation + }); + + return result.toUIMessageStreamResponse(); +} +``` + +If you do not pass `abortSignal` to `streamText`, the LLM call will continue running in the background even after the user cancels. Always forward it when possible. + +### Stream Recovery + +When a Durable Object is evicted mid-stream (code update, inactivity timeout, resource limit), the LLM connection is severed permanently and the in-memory streaming state is lost. `chatRecovery` wraps each chat turn in a [`runFiber()`](./durable-execution.md), providing automatic `keepAlive` during streaming and a recovery hook on restart. + +```typescript +export class ChatAgent extends AIChatAgent { + override chatRecovery = true; +} +``` + +When enabled, every `onChatMessage` call runs inside a fiber. If the agent is evicted mid-stream, the fiber row survives in SQLite. On the next activation, the framework detects the interrupted fiber, reconstructs the partial response from buffered stream chunks, and calls `onChatRecovery`. + +#### `onChatRecovery` + +Override to implement provider-specific recovery. The default behavior persists the partial response and schedules a continuation via `continueLastTurn()`. + +```typescript +export class ChatAgent extends AIChatAgent { + override chatRecovery = true; + + override async onChatRecovery( + ctx: ChatRecoveryContext + ): Promise { + // Inspect what was generated before eviction + console.log(`Recovered ${ctx.partialText.length} chars of partial text`); + + // Default: persist partial + schedule continuation + return {}; + } +} +``` + +**`ChatRecoveryContext`:** + +| Field | Type | Description | +| ----------------- | -------------------------------------- | --------------------------------------------------------------------- | +| `streamId` | `string` | ID of the interrupted stream | +| `requestId` | `string` | ID of the original chat request | +| `partialText` | `string` | Text generated before eviction | +| `partialParts` | `MessagePart[]` | Message parts (text, reasoning, tool calls) generated before eviction | +| `recoveryData` | `unknown \| null` | Data from `this.stash()` — entirely user-controlled | +| `messages` | `ChatMessage[]` | Full conversation history | +| `lastBody` | `Record \| undefined` | The original request body | +| `lastClientTools` | `ClientToolSchema[] \| undefined` | Client tool schemas from the original request | + +**`ChatRecoveryOptions`:** + +| Field | Default | Description | +| ---------- | ------- | ------------------------------------------------- | +| `persist` | `true` | Save the partial response as an assistant message | +| `continue` | `true` | Schedule a continuation via `continueLastTurn()` | + +Common return values: + +- `{}` — persist partial + auto-continue (default, works with providers that support assistant prefill) +- `{ continue: false }` — persist partial but do not auto-continue (handle continuation yourself) +- `{ persist: false, continue: false }` — handle everything yourself (e.g., retrieve a completed response from the provider) + +#### `continueLastTurn` + +Appends to the last assistant message by re-calling `onChatMessage` with the saved request body. The response is streamed as a continuation — appended to the existing assistant message, not a new one. No synthetic user message is created. + +```typescript +protected continueLastTurn(body?: Record): Promise; +``` + +Called automatically by the default recovery path. Can also be called manually from scheduled callbacks or other entry points. The optional `body` parameter merges with the saved `_lastBody`. + +#### Stashing recovery data + +Use `this.stash()` inside `onChatMessage` to persist provider-specific data for recovery. The stash is stored in the fiber's SQLite row, separate from agent state, and available as `ctx.recoveryData` in `onChatRecovery`. + +```typescript +async onChatMessage(_onFinish, options) { + const result = streamText({ + model: openai("gpt-5.4"), + messages: await convertToModelMessages(this.messages), + providerOptions: { openai: { store: true } }, + includeRawChunks: true, + onChunk: ({ chunk }) => { + if (chunk.type === "raw") { + const raw = chunk.rawValue as { type?: string; response?: { id?: string } }; + if (raw?.type === "response.created" && raw.response?.id) { + this.stash({ responseId: raw.response.id }); + } + } + } + }); + return result.toUIMessageStreamResponse(); +} +``` + +#### Recovery strategies by provider + +The right strategy depends on whether the provider supports assistant prefill and whether the response continues server-side after disconnection: + +| Provider | Strategy | Token cost | +| ---------------------- | ------------------------------------------------------------ | ---------- | +| Workers AI | `continueLastTurn()` — model continues via assistant prefill | Low | +| OpenAI (Responses API) | Retrieve completed response by ID — zero wasted tokens | Zero | +| Anthropic | Persist partial, send a synthetic user message to continue | Medium | + +For a complete multi-provider implementation, see the [`forever-chat` example](../experimental/forever-chat/) and the [`forever.md` design doc](../experimental/forever.md). For how chat recovery fits into the broader long-running agents story, see [Long-Running Agents: Recovering interrupted LLM streams](./long-running-agents.md#recovering-interrupted-llm-streams). + +## Client API + +### `useAgentChat` + +React hook that connects to an `AIChatAgent` over WebSocket. Wraps the AI SDK's `useChat` with a native WebSocket transport. + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "ChatAgent" }); + const { + messages, + sendMessage, + clearHistory, + addToolOutput, + addToolApprovalResponse, + setMessages, + status, + isServerStreaming, + isStreaming + } = useAgentChat({ agent }); + + // ... +} +``` + +### Options + +| Option | Type | Default | Description | +| ----------------------------- | --------------------------------------------- | -------- | ------------------------------------------------------------------------------------------------------------------------ | +| `agent` | `ReturnType` | Required | Agent connection from `useAgent` | +| `onToolCall` | `({ toolCall, addToolOutput }) => void` | — | Handle client-side tool execution | +| `autoContinueAfterToolResult` | `boolean` | `true` | Auto-continue conversation after client tool results and approvals | +| `resume` | `boolean` | `true` | Enable automatic stream resumption on reconnect | +| `body` | `object \| () => object` | — | Custom data sent with every request | +| `prepareSendMessagesRequest` | `(options) => { body?, headers? }` | — | Advanced per-request customization | +| `tools` | `Record` | — | Dynamic client-defined tools for SDK/platform use cases. Schemas are sent to the server automatically | +| `getInitialMessages` | `(options) => Promise` or `null` | — | Custom initial message loader. Set to `null` to skip the HTTP fetch entirely (useful when providing `messages` directly) | + +### Return Values + +| Property | Type | Description | +| ------------------------- | ---------------------------------- | -------------------------------------------------------------------------- | +| `messages` | `UIMessage[]` | Current conversation messages | +| `sendMessage` | `(message) => void` | Send a message | +| `clearHistory` | `() => void` | Clear conversation (client and server) | +| `addToolOutput` | `({ toolCallId, output }) => void` | Provide output for a client-side tool | +| `addToolApprovalResponse` | `({ id, approved }) => void` | Approve or reject a tool requiring approval | +| `setMessages` | `(messages \| updater) => void` | Set messages directly (syncs to server) | +| `status` | `string` | `"idle"`, `"submitted"`, `"streaming"`, or `"error"` | +| `isServerStreaming` | `boolean` | `true` when a server-initiated stream is active (e.g. from `saveMessages`) | +| `isStreaming` | `boolean` | `true` when any stream is active (client or server-initiated) | + +## Tools + +`AIChatAgent` supports three tool patterns, all using the AI SDK's `tool()` function: + +| Pattern | Where it runs | When to use | +| ----------- | ---------------------------- | --------------------------------------------- | +| Server-side | Server (automatic) | API calls, database queries, computations | +| Client-side | Browser (via `onToolCall`) | Geolocation, clipboard, camera, local storage | +| Approval | Server (after user approval) | Payments, deletions, external actions | + +### Server-Side Tools + +Tools with an `execute` function run automatically on the server: + +```typescript +import { streamText, convertToModelMessages, tool, stepCountIs } from "ai"; +import { z } from "zod"; + +async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages), + tools: { + getWeather: tool({ + description: "Get weather for a city", + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }) => { + const data = await fetchWeather(city); + return { temperature: data.temp, condition: data.condition }; + } + }) + }, + stopWhen: stepCountIs(5) + }); + + return result.toUIMessageStreamResponse(); +} +``` + +### Client-Side Tools + +Define a tool on the server without `execute`, then handle it on the client with `onToolCall`. Use this for tools that need browser APIs: + +**Server:** + +```typescript +tools: { + getLocation: tool({ + description: "Get the user's location from the browser", + inputSchema: z.object({}) + // No execute — the client handles it + }); +} +``` + +**Client:** + +```tsx +const { messages, sendMessage } = useAgentChat({ + agent, + onToolCall: async ({ toolCall, addToolOutput }) => { + if (toolCall.toolName === "getLocation") { + const pos = await new Promise((resolve, reject) => + navigator.geolocation.getCurrentPosition(resolve, reject) + ); + addToolOutput({ + toolCallId: toolCall.toolCallId, + output: { lat: pos.coords.latitude, lng: pos.coords.longitude } + }); + } + } +}); +``` + +When the LLM invokes `getLocation`, the stream pauses. The `onToolCall` callback fires, your code provides the output, and the conversation continues. + +### Dynamic Client Tools (SDK/Platform Pattern) + +For SDKs and platforms where tools are defined dynamically by the embedding application at runtime, use the `tools` option on `useAgentChat` and `createToolsFromClientSchemas()` on the server: + +**Server:** + +```typescript +import { createToolsFromClientSchemas } from "@cloudflare/ai-chat"; + +async onChatMessage(_onFinish, options) { + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages), + tools: createToolsFromClientSchemas(options?.clientTools) + }); + return result.toUIMessageStreamResponse(); +} +``` + +**Client:** + +```tsx +import { useAgentChat, type AITool } from "@cloudflare/ai-chat/react"; + +const tools: Record = { + getPageTitle: { + description: "Get the current page title", + parameters: { type: "object", properties: {} }, + execute: async () => ({ title: document.title }) + } +}; + +const { messages, sendMessage } = useAgentChat({ + agent, + tools, + onToolCall: async ({ toolCall, addToolOutput }) => { + const tool = tools[toolCall.toolName]; + if (tool?.execute) { + const output = await tool.execute(toolCall.input); + addToolOutput({ toolCallId: toolCall.toolCallId, output }); + } + } +}); +``` + +For most apps, server-side tools with `tool()` and `onToolCall` are simpler and provide full Zod type safety. Use dynamic client tools when the server does not know the tool surface at deploy time. + +### Tool Approval (Human-in-the-Loop) + +Use `needsApproval` for tools that require user confirmation before executing: + +**Server:** + +```typescript +tools: { + processPayment: tool({ + description: "Process a payment", + inputSchema: z.object({ + amount: z.number(), + recipient: z.string() + }), + needsApproval: async ({ amount }) => amount > 100, + execute: async ({ amount, recipient }) => charge(amount, recipient) + }); +} +``` + +**Client:** + +```tsx +import { isToolUIPart, getToolName } from "ai"; + +const { messages, addToolApprovalResponse } = useAgentChat({ agent }); + +// Render pending approvals from message parts +{ + messages.map((msg) => + msg.parts + .filter( + (part) => + isToolUIPart(part) && + "approval" in part && + part.state === "approval-requested" + ) + .map((part) => ( +
+

Approve {getToolName(part)}?

+ + +
+ )) + ); +} +``` + +When denied, the tool part transitions to `output-denied`. You can also use `addToolOutput` with `state: "output-error"` for custom denial messages — see [Human in the Loop](./human-in-the-loop.md) for details. + +## Custom Request Data + +Include custom data with every chat request using the `body` option: + +```tsx +const { messages, sendMessage } = useAgentChat({ + agent, + body: { + timezone: Intl.DateTimeFormat().resolvedOptions().timeZone, + userId: currentUser.id + } +}); +``` + +For dynamic values, use a function: + +```tsx +body: () => ({ + token: getAuthToken(), + timestamp: Date.now() +}); +``` + +Access these fields on the server: + +```typescript +async onChatMessage(_onFinish, options) { + const { timezone, userId } = options?.body ?? {}; + // ... +} +``` + +For advanced per-request customization (custom headers, different body per request), use `prepareSendMessagesRequest`: + +```tsx +const { messages, sendMessage } = useAgentChat({ + agent, + prepareSendMessagesRequest: async ({ messages, trigger }) => ({ + headers: { Authorization: `Bearer ${await getToken()}` }, + body: { requestedAt: Date.now() } + }) +}); +``` + +## Data Parts + +Data parts let you attach typed JSON to messages alongside text — progress indicators, source citations, token usage, or any structured data your UI needs. + +### Writing Data Parts (Server) + +Use `createUIMessageStream` with `writer.write()` to send data parts from the server: + +```ts +import { + streamText, + convertToModelMessages, + createUIMessageStream, + createUIMessageStreamResponse +} from "ai"; + +export class ChatAgent extends AIChatAgent { + async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const stream = createUIMessageStream({ + execute: async ({ writer }) => { + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages) + }); + + // Merge the LLM stream + writer.merge(result.toUIMessageStream()); + + // Write a data part — persisted to message.parts + writer.write({ + type: "data-sources", + id: "src-1", + data: { query: "agents", status: "searching", results: [] } + }); + + // Later: update the same part in-place (same type + id) + writer.write({ + type: "data-sources", + id: "src-1", + data: { + query: "agents", + status: "found", + results: ["Agents SDK docs", "Durable Objects guide"] + } + }); + } + }); + + return createUIMessageStreamResponse({ stream }); + } +} +``` + +### Three Patterns + +| Pattern | How | Persisted? | Use case | +| ------------------ | ------------------------------------------------ | ---------- | ------------------------------------- | +| **Reconciliation** | Same `type` + `id` → updates in-place | Yes | Progressive state (searching → found) | +| **Append** | No `id`, or different `id` → appends | Yes | Log entries, multiple citations | +| **Transient** | `transient: true` → not added to `message.parts` | No | Ephemeral status (thinking indicator) | + +Transient parts are broadcast to connected clients in real time but excluded from SQLite persistence and `message.parts`. Use the `onData` callback to consume them. + +### Reading Data Parts (Client) + +Non-transient data parts appear in `message.parts`. Use the `UIMessage` generic to type them: + +```ts +import { useAgentChat } from "@cloudflare/ai-chat/react"; +import type { UIMessage } from "ai"; + +type ChatMessage = UIMessage< + unknown, + { + sources: { query: string; status: string; results: string[] }; + usage: { model: string; inputTokens: number; outputTokens: number }; + } +>; + +const { messages } = useAgentChat({ agent }); + +// Typed access — no casts needed +for (const msg of messages) { + for (const part of msg.parts) { + if (part.type === "data-sources") { + console.log(part.data.results); // string[] + } + } +} +``` + +### Transient Parts with `onData` + +Transient data parts are not in `message.parts`. Use the `onData` callback instead: + +```ts +const [thinking, setThinking] = useState(false); + +const { messages } = useAgentChat({ + agent, + onData(part) { + if (part.type === "data-thinking") { + setThinking(true); + } + } +}); +``` + +On the server, write transient parts with `transient: true`: + +```ts +writer.write({ + transient: true, + type: "data-thinking", + data: { model: "glm-4.7-flash", startedAt: new Date().toISOString() } +}); +``` + +`onData` fires on all code paths — new messages, stream resumption, and cross-tab broadcasts. + +## Resumable Streaming + +Streams automatically resume when a client disconnects and reconnects. No configuration is needed — it works out of the box. + +When streaming is active: + +1. All chunks are buffered in SQLite as they are generated +2. If the client disconnects, the server continues streaming and buffering +3. When the client reconnects, it receives all buffered chunks and resumes live streaming + +Disable with `resume: false`: + +```tsx +const { messages } = useAgentChat({ agent, resume: false }); +``` + +For more details, see [Resumable Streaming](./resumable-streaming.md). + +## Storage Management + +### Row Size Protection + +SQLite rows have a maximum size of 2 MB. When a message approaches this limit (for example, a tool returning a very large output), `AIChatAgent` automatically compacts the message: + +1. **Tool output compaction** — Large tool outputs are replaced with an LLM-friendly summary that instructs the model to suggest re-running the tool +2. **Text truncation** — If the message is still too large after tool compaction, text parts are truncated with a note + +Compacted messages include `metadata.compactedToolOutputs` so clients can detect and display this gracefully. + +### `sanitizeMessageForPersistence` + +Override this method to transform messages before they are written to SQLite. It runs after the built-in sanitization (OpenAI metadata stripping, Anthropic provider-executed tool payload truncation, empty reasoning part removal), so built-in cleanup is never bypassed. + +The default implementation returns the message unchanged. + +```typescript +import type { UIMessage } from "ai"; + +export class ChatAgent extends AIChatAgent { + protected sanitizeMessageForPersistence(message: UIMessage): UIMessage { + return { + ...message, + parts: message.parts.map((part) => { + // Strip large tool outputs you do not need to persist + if ( + "output" in part && + typeof part.output === "string" && + part.output.length > 2000 + ) { + return { ...part, output: "[redacted — too large to store]" }; + } + return part; + }) + }; + } +} +``` + +Built-in sanitization already handles: + +- **OpenAI** — strips ephemeral `itemId` and `reasoningEncryptedContent` from `providerMetadata` +- **Anthropic** — truncates large strings in `input`/`output` of provider-executed tool parts (code execution, text editor) +- **Reasoning** — removes empty reasoning parts left after metadata stripping + +Use `sanitizeMessageForPersistence` for anything the built-in logic does not cover, such as redacting sensitive fields or trimming domain-specific tool outputs. + +### Controlling LLM Context vs Storage + +Storage (`maxPersistedMessages`) and LLM context are independent: + +| Concern | Control | Scope | +| ------------------------------- | --------------------------------- | ----------- | +| How many messages SQLite stores | `maxPersistedMessages` | Persistence | +| What the model sees | `pruneMessages()` | LLM context | +| Custom pre-persist transforms | `sanitizeMessageForPersistence()` | Per-message | +| Row size limits | Automatic compaction | Per-message | + +```typescript +export class ChatAgent extends AIChatAgent { + maxPersistedMessages = 200; // Storage limit + + async onChatMessage() { + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: pruneMessages({ + // LLM context limit + messages: await convertToModelMessages(this.messages), + reasoning: "before-last-message", + toolCalls: "before-last-2-messages" + }) + }); + + return result.toUIMessageStreamResponse(); + } +} +``` + +## Using Different AI Providers + +`AIChatAgent` works with any AI SDK-compatible provider. The server code determines which model to use — the client does not need to change. + +### Workers AI (Cloudflare) + +```typescript +import { createWorkersAI } from "workers-ai-provider"; + +const workersai = createWorkersAI({ binding: this.env.AI }); +const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages) +}); +``` + +### OpenAI + +```typescript +import { createOpenAI } from "@ai-sdk/openai"; + +const openai = createOpenAI({ apiKey: this.env.OPENAI_API_KEY }); +const result = streamText({ + model: openai.chat("gpt-4o"), + messages: await convertToModelMessages(this.messages) +}); +``` + +### Anthropic + +```typescript +import { createAnthropic } from "@ai-sdk/anthropic"; + +const anthropic = createAnthropic({ apiKey: this.env.ANTHROPIC_API_KEY }); +const result = streamText({ + model: anthropic("claude-sonnet-4-20250514"), + messages: await convertToModelMessages(this.messages) +}); +``` + +## Advanced Patterns + +Since `onChatMessage` gives you full control over the `streamText` call, you can use any AI SDK feature directly. The patterns below all work out of the box — no special `AIChatAgent` configuration is needed. + +### Dynamic Model and Tool Control + +Use [`prepareStep`](https://ai-sdk.dev/docs/agents/loop-control) to change the model, available tools, or system prompt between steps in a multi-step agent loop: + +```typescript +import { streamText, convertToModelMessages, tool, stepCountIs } from "ai"; +import { z } from "zod"; + +async onChatMessage() { + const result = streamText({ + model: cheapModel, // Default model for simple steps + messages: await convertToModelMessages(this.messages), + tools: { + search: searchTool, + analyze: analyzeTool, + summarize: summarizeTool + }, + stopWhen: stepCountIs(10), + prepareStep: async ({ stepNumber, messages }) => { + // Phase 1: Search (steps 0-2) + if (stepNumber <= 2) { + return { + activeTools: ["search"], + toolChoice: "required" // Force tool use + }; + } + + // Phase 2: Analyze with a stronger model (steps 3-5) + if (stepNumber <= 5) { + return { + model: expensiveModel, + activeTools: ["analyze"] + }; + } + + // Phase 3: Summarize + return { activeTools: ["summarize"] }; + } + }); + + return result.toUIMessageStreamResponse(); +} +``` + +`prepareStep` runs before each step and can return overrides for `model`, `activeTools`, `toolChoice`, `system`, and `messages`. Use it to: + +- **Switch models** — use a cheap model for simple steps, escalate for reasoning +- **Phase tools** — restrict which tools are available at each step +- **Manage context** — prune or transform messages to stay within token limits +- **Force tool calls** — use `toolChoice: { type: "tool", toolName: "search" }` to require a specific tool + +### Language Model Middleware + +Use [`wrapLanguageModel`](https://ai-sdk.dev/docs/ai-sdk-core/middleware) to add guardrails, RAG, caching, or logging without modifying your chat logic: + +```typescript +import { streamText, convertToModelMessages, wrapLanguageModel } from "ai"; +import type { LanguageModelV3Middleware } from "@ai-sdk/provider"; + +const guardrailMiddleware: LanguageModelV3Middleware = { + wrapGenerate: async ({ doGenerate }) => { + const { text, ...rest } = await doGenerate(); + // Filter PII or sensitive content from the response + const cleaned = text?.replace(/\b\d{3}-\d{2}-\d{4}\b/g, "[REDACTED]"); + return { text: cleaned, ...rest }; + } +}; + +async onChatMessage() { + const model = wrapLanguageModel({ + model: baseModel, + middleware: [guardrailMiddleware] + }); + + const result = streamText({ + model, + messages: await convertToModelMessages(this.messages) + }); + + return result.toUIMessageStreamResponse(); +} +``` + +The AI SDK includes built-in middlewares: + +- `extractReasoningMiddleware` — surface chain-of-thought from models like DeepSeek R1 +- `defaultSettingsMiddleware` — apply default temperature, max tokens, etc. +- `simulateStreamingMiddleware` — add streaming to non-streaming models + +Multiple middlewares compose in order: `middleware: [first, second]` applies as `first(second(model))`. + +### Structured Output + +Use [`generateObject`](https://ai-sdk.dev/docs/ai-sdk-core/generating-structured-data) inside tools for structured data extraction: + +```typescript +import { + streamText, generateObject, convertToModelMessages, tool, stepCountIs +} from "ai"; +import { z } from "zod"; + +async onChatMessage() { + const result = streamText({ + model: myModel, + messages: await convertToModelMessages(this.messages), + tools: { + extractContactInfo: tool({ + description: "Extract structured contact information from the conversation", + inputSchema: z.object({ + text: z.string().describe("The text to extract contact info from") + }), + execute: async ({ text }) => { + const { object } = await generateObject({ + model: myModel, + schema: z.object({ + name: z.string(), + email: z.string().email(), + phone: z.string().optional() + }), + prompt: `Extract contact information from: ${text}` + }); + return object; + } + }) + }, + stopWhen: stepCountIs(5) + }); + + return result.toUIMessageStreamResponse(); +} +``` + +### Subagent Delegation + +Tools can delegate work to focused sub-calls with their own context. Use [`ToolLoopAgent`](https://ai-sdk.dev/docs/reference/ai-sdk-core/tool-loop-agent) to define a reusable agent, then call it from a tool's `execute`: + +```typescript +import { + ToolLoopAgent, streamText, convertToModelMessages, tool, stepCountIs +} from "ai"; +import { z } from "zod"; + +// Define a reusable research agent with its own tools and instructions +const researchAgent = new ToolLoopAgent({ + model: researchModel, + instructions: "You are a research assistant. Be thorough and cite sources.", + tools: { webSearch: webSearchTool }, + stopWhen: stepCountIs(10) +}); + +async onChatMessage() { + const result = streamText({ + model: orchestratorModel, + messages: await convertToModelMessages(this.messages), + tools: { + deepResearch: tool({ + description: "Research a topic in depth", + inputSchema: z.object({ + topic: z.string().describe("The topic to research") + }), + execute: async ({ topic }) => { + const { text } = await researchAgent.generate({ prompt: topic }); + return { summary: text }; + } + }) + }, + stopWhen: stepCountIs(5) + }); + + return result.toUIMessageStreamResponse(); +} +``` + +The research agent runs in its own context — its token budget is separate from the orchestrator's. Only the summary goes back to the parent model. + +`ToolLoopAgent` is best suited for subagents, not as a replacement for `streamText` in `onChatMessage` itself. The main `onChatMessage` benefits from direct access to `this.env`, `this.messages`, and `options.body` — things that a pre-configured `ToolLoopAgent` instance cannot reference. + +#### Streaming progress with preliminary results + +By default, a tool part appears as loading until `execute` returns. Use an async generator (`async function*`) to stream progress updates to the client while the tool is still working: + +```typescript +deepResearch: tool({ + description: "Research a topic in depth", + inputSchema: z.object({ + topic: z.string().describe("The topic to research") + }), + async *execute({ topic }) { + // Preliminary result — the client sees "searching" immediately + yield { status: "searching", topic, summary: undefined }; + + const { text } = await researchAgent.generate({ prompt: topic }); + + // Final result — sent to the model for its next step + yield { status: "done", topic, summary: text }; + } +}); +``` + +Each `yield` updates the tool part on the client in real-time (with `preliminary: true`). The last yielded value becomes the final output that the model sees. This is useful for long-running tools where you want to show status (searching, analyzing, summarizing) as the work progresses. + +This pattern is useful when: + +- A task requires exploring large amounts of information that would bloat the main context +- You want to show real-time progress for long-running tools +- You want to parallelize independent research (multiple tool calls run concurrently) +- You need different models or system prompts for different subtasks + +For more, see the [AI SDK Agents docs](https://ai-sdk.dev/docs/agents/overview), [Subagents](https://ai-sdk.dev/docs/agents/subagents), and [Preliminary Tool Results](https://ai-sdk.dev/docs/ai-sdk-core/tools-and-tool-calling#preliminary-tool-results). + +## Multi-Client Sync + +When multiple clients connect to the same agent instance, messages are automatically broadcast to all connections. If one client sends a message, all other connected clients receive the updated message list. + +``` +Client A ──── sendMessage("Hello") ────▶ AIChatAgent + │ + persist + stream + │ +Client A ◀── CF_AGENT_USE_CHAT_RESPONSE ──────┤ +Client B ◀── CF_AGENT_CHAT_MESSAGES ──────────┘ +``` + +The originating client receives the streaming response. All other clients receive the final messages via a `CF_AGENT_CHAT_MESSAGES` broadcast. + +## API Reference + +### Exports + +| Import path | Exports | +| --------------------------- | ------------------------------------------------------------------------------------------- | +| `@cloudflare/ai-chat` | `AIChatAgent`, `createToolsFromClientSchemas`, `ChatRecoveryContext`, `ChatRecoveryOptions` | +| `@cloudflare/ai-chat/react` | `useAgentChat` | +| `@cloudflare/ai-chat/types` | `MessageType`, `OutgoingMessage`, `IncomingMessage` | + +### WebSocket Protocol + +The chat protocol uses typed JSON messages over WebSocket: + +| Message | Direction | Purpose | +| -------------------------------- | --------------- | --------------------------- | +| `CF_AGENT_USE_CHAT_REQUEST` | Client → Server | Send a chat message | +| `CF_AGENT_USE_CHAT_RESPONSE` | Server → Client | Stream response chunks | +| `CF_AGENT_CHAT_MESSAGES` | Server → Client | Broadcast updated messages | +| `CF_AGENT_CHAT_CLEAR` | Bidirectional | Clear conversation | +| `CF_AGENT_CHAT_REQUEST_CANCEL` | Client → Server | Cancel active stream | +| `CF_AGENT_TOOL_RESULT` | Client → Server | Provide tool output | +| `CF_AGENT_TOOL_APPROVAL` | Client → Server | Approve or reject a tool | +| `CF_AGENT_MESSAGE_UPDATED` | Server → Client | Notify of message update | +| `CF_AGENT_STREAM_RESUMING` | Server → Client | Notify of stream resumption | +| `CF_AGENT_STREAM_RESUME_REQUEST` | Client → Server | Request stream resume check | + +## Examples + +- [AI Chat Example](../examples/ai-chat/) — Modern example with server tools, client tools, and approval +- [Dynamic Tools](../examples/dynamic-tools/) — SDK/platform pattern with dynamic client-defined tools +- [Resumable Stream Chat](../examples/resumable-stream-chat/) — Automatic stream resumption demo +- [Human in the Loop Guide](../guides/human-in-the-loop/) — Tool approval with `needsApproval` and `onToolCall` +- [Playground](../examples/playground/) — Kitchen-sink demo of all SDK features + +## Related Docs + +- [Client SDK](./client-sdk.md) — `useAgent` hook and `AgentClient` class +- [Durable Execution](./durable-execution.md) — `runFiber()`, `stash()`, and crash recovery +- [Long-Running Agents](./long-running-agents.md) — lifecycle, recovery patterns, and provider-specific strategies +- [Human in the Loop](./human-in-the-loop.md) — Approval flows and manual intervention patterns +- [Resumable Streaming](./resumable-streaming.md) — How stream resumption works +- [Client Tools Continuation](./client-tools-continuation.md) — Advanced client-side tool patterns +- [Migration to AI SDK v6](./migration-to-ai-sdk-v6.md) — Upgrading from AI SDK v5 diff --git a/docs/client-sdk.md b/docs/client-sdk.md new file mode 100644 index 0000000000..7d6fee8339 --- /dev/null +++ b/docs/client-sdk.md @@ -0,0 +1,520 @@ +# Client SDK + +Connect to agents from any JavaScript runtime — browsers, Node.js, Deno, Bun, or edge functions — using WebSockets or HTTP. The SDK provides real-time state synchronization, RPC method calls, and streaming responses. + +## Overview + +The client SDK offers two ways to connect with a websocket connection, and one way to make HTTP requests. + +| Client | Use Case | +| ------------- | ----------------------------------------------------------- | +| `useAgent` | React hook with automatic reconnection and state management | +| `AgentClient` | Vanilla JavaScript/TypeScript class for any environment | +| `agentFetch` | HTTP requests when WebSocket isn't needed | + +All clients provide: + +- **Bidirectional state sync** - Push and receive state updates in real-time +- **RPC calls** - Call agent methods with typed arguments and return values +- **Streaming** - Handle chunked responses for AI completions +- **Auto-reconnection** - Built on [PartySocket](https://docs.partykit.io/reference/partysocket-api/) for reliable connections + +## Quick Start + +### React + +```tsx +import { useAgent } from "agents/react"; + +function Chat() { + const agent = useAgent({ + agent: "ChatAgent", + name: "room-123" + }); + + const sendMessage = async () => { + const response = await agent.call("sendMessage", ["Hello!"]); + console.log("Response:", response); + }; + + return ( +
+

Messages: {agent.state?.messageCount ?? 0}

+ +
+ ); +} +``` + +### Vanilla JavaScript + +```typescript +import { AgentClient } from "agents/client"; + +const client = new AgentClient({ + agent: "ChatAgent", + name: "room-123", + host: "your-worker.your-subdomain.workers.dev" +}); + +await client.ready; + +// Read state directly +console.log("Current state:", client.state); + +// Call a method +const response = await client.call("sendMessage", ["Hello!"]); +``` + +## Connecting to Agents + +### Agent Naming + +The `agent` parameter is your agent class name. It's automatically converted from camelCase to kebab-case for the URL: + +```typescript +// These are equivalent: +useAgent({ agent: "ChatAgent" }); // → /agents/chat-agent/... +useAgent({ agent: "MyCustomAgent" }); // → /agents/my-custom-agent/... +useAgent({ agent: "LOUD_AGENT" }); // → /agents/loud-agent/... +``` + +### Instance Names + +The `name` parameter identifies a specific agent instance. If omitted, defaults to `"default"`: + +```typescript +// Connect to a specific chat room +useAgent({ agent: "ChatAgent", name: "room-123" }); + +// Connect to a user's personal agent +useAgent({ agent: "UserAgent", name: userId }); + +// Uses "default" instance +useAgent({ agent: "ChatAgent" }); +``` + +### Connection Options + +Both `useAgent` and `AgentClient` accept PartySocket options: + +```typescript +useAgent({ + agent: "ChatAgent", + name: "room-123", + + // Connection settings + host: "my-worker.workers.dev", // Custom host (defaults to current origin) + path: "/custom/path", // Custom path prefix + + // Query parameters (sent on connection) + query: { + token: "abc123", + version: "2" + }, + + // Event handlers + onOpen: () => console.log("Connected"), + onClose: () => console.log("Disconnected"), + onError: (error) => console.error("Error:", error) +}); +``` + +### Async Query Parameters + +For authentication tokens or other async data, pass a function that returns a Promise: + +```typescript +useAgent({ + agent: "ChatAgent", + name: "room-123", + + // Async query - called before connecting + query: async () => { + const token = await getAuthToken(); + return { token }; + }, + + // Dependencies that trigger re-fetching the query + queryDeps: [userId], + + // Cache TTL for the query result (default: 5 minutes) + cacheTtl: 60 * 1000 // 1 minute +}); +``` + +The query function is cached and only re-called when: + +- `queryDeps` change +- `cacheTtl` expires +- The component remounts + +## State Synchronization + +Agents can maintain state that syncs bidirectionally with all connected clients. Both `useAgent` and `AgentClient` expose a `state` property that tracks the current agent state. + +### Reading State + +```tsx +const agent = useAgent({ + agent: "GameAgent", + name: "game-123" +}); + +// Read state directly — reactive in React (re-renders on change) +return
Score: {agent.state?.score}
; +``` + +`agent.state` starts as `undefined` and is populated when the server sends state on connect (from the agent's `initialState`). Use optional chaining for safe access. + +### Pushing State Updates + +```typescript +// Update the agent's state from the client +agent.setState({ score: 100, level: 5 }); + +// Spread existing state for partial updates +agent.setState({ ...agent.state, score: agent.state.score + 10 }); +``` + +When you call `setState()`: + +1. The state is sent to the agent over WebSocket +2. The agent's `onStateChanged()` method is called +3. The agent broadcasts the new state to all connected clients +4. `agent.state` updates on the next render (React) or immediately (`AgentClient`) + +### Listening for State Changes + +For side effects when state changes, use the `onStateUpdate` callback: + +```typescript +const agent = useAgent({ + agent: "GameAgent", + name: "game-123", + onStateUpdate: (state, source) => { + // source: "server" (agent pushed) or "client" (you pushed) + console.log(`State updated from ${source}:`, state); + } +}); +``` + +> For most use cases, reading `agent.state` directly is simpler than tracking state manually with `onStateUpdate`. Use `onStateUpdate` when you need to trigger side effects on state changes. + +### State Flow + +``` +┌─────────┐ ┌─────────┐ +│ Client │ ── setState() ────▶ │ Agent │ +│ │ │ │ +│ .state │ ◀── state update ── │ │ +└─────────┘ (broadcast) └─────────┘ +``` + +## Calling Agent Methods (RPC) + +Call methods on your agent that are decorated with `@callable()`. + +> **Note:** The `@callable()` decorator is only required for methods called from external runtimes (browsers, other services). When calling from within the same Worker, you can use standard [Durable Object RPC](https://developers.cloudflare.com/durable-objects/best-practices/create-durable-object-stubs-and-send-requests/#invoke-rpc-methods) directly on the stub without the decorator. + +### Using call() + +```typescript +// Basic call +const result = await agent.call("getUser", [userId]); + +// Call with multiple arguments +const result = await agent.call("createPost", [title, content, tags]); + +// Call with no arguments +const result = await agent.call("getStats"); +``` + +### Using the Stub Proxy + +The `stub` property provides a cleaner syntax for method calls: + +```typescript +// Instead of: +const user = await agent.call("getUser", ["user-123"]); + +// You can write: +const user = await agent.stub.getUser("user-123"); + +// Multiple arguments work naturally: +const post = await agent.stub.createPost(title, content, tags); +``` + +### TypeScript Integration + +For full type safety, pass your Agent class as a type parameter: + +```typescript +import type { MyAgent } from "./agents/my-agent"; + +const agent = useAgent({ + agent: "MyAgent", + name: "instance-1" +}); + +// Now stub methods are fully typed! +const result = await agent.stub.processData({ input: "test" }); +// ^? Awaited> +``` + +### Streaming Responses + +For methods that return `StreamingResponse`, handle chunks as they arrive: + +```typescript +// Agent-side: +@callable() +async generateText(prompt: string) { + return new StreamingResponse(async (stream) => { + for await (const chunk of llm.stream(prompt)) { + await stream.write(chunk); + } + }); +} + +// Client-side: +await agent.call("generateText", [prompt], { + onChunk: (chunk) => { + // Called for each chunk + appendToOutput(chunk); + }, + onDone: (finalResult) => { + // Called when stream completes + console.log("Complete:", finalResult); + }, + onError: (error) => { + // Called if streaming fails + console.error("Stream error:", error); + } +}); +``` + +## HTTP Requests with agentFetch + +For one-off requests without maintaining a WebSocket connection: + +```typescript +import { agentFetch } from "agents/client"; + +// GET request +const response = await agentFetch({ + agent: "DataAgent", + name: "instance-1", + host: "my-worker.workers.dev" +}); + +const data = await response.json(); + +// POST request with body +const response = await agentFetch( + { + agent: "DataAgent", + name: "instance-1", + host: "my-worker.workers.dev" + }, + { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ action: "process" }) + } +); +``` + +**When to use `agentFetch` vs WebSocket:** + +| Use `agentFetch` | Use `useAgent`/`AgentClient` | +| ------------------------------- | ---------------------------- | +| One-time requests | Real-time updates needed | +| Server-to-server calls | Bidirectional communication | +| Simple REST-style API | State synchronization | +| No persistent connection needed | Multiple RPC calls | + +## React Hook Reference + +### UseAgentOptions + +```typescript +type UseAgentOptions = { + // Required + agent: string; // Agent class name + + // Optional + name?: string; // Instance name (default: "default") + host?: string; // Custom host + path?: string; // Custom path prefix + + // Query parameters + query?: + | Record + | (() => Promise>); + queryDeps?: unknown[]; // Dependencies for async query + cacheTtl?: number; // Query cache TTL in ms (default: 5 min) + + // Callbacks + onStateUpdate?: (state: State, source: "server" | "client") => void; + onMcpUpdate?: (mcpServers: MCPServersState) => void; + onOpen?: () => void; + onClose?: () => void; + onError?: (error: Event) => void; + onMessage?: (message: MessageEvent) => void; +}; +``` + +### Return Value + +```typescript +const agent = useAgent(options); + +agent.state; // State | undefined - Current agent state (reactive) +agent.agent; // string - Kebab-case agent name +agent.name; // string - Instance name +agent.setState(state); // void - Push state to agent +agent.call(method, args?, streamOptions?); // Promise - Call agent method +agent.stub; // Proxy - Typed method calls +agent.send(data); // void - Send raw WebSocket message +agent.close(); // void - Close connection +agent.reconnect(); // void - Force reconnection +``` + +## Vanilla JS Reference + +### AgentClientOptions + +```typescript +type AgentClientOptions = { + // Required + agent: string; // Agent class name + host: string; // Worker host + + // Optional + name?: string; // Instance name (default: "default") + path?: string; // Custom path prefix + query?: Record; + + // Callbacks + onStateUpdate?: (state: State, source: "server" | "client") => void; +}; +``` + +### AgentClient Methods + +```typescript +const client = new AgentClient(options); + +client.state; // State | undefined - Current agent state +client.agent; // string - Kebab-case agent name +client.name; // string - Instance name +client.setState(state); // void - Push state to agent +client.call(method, args?, streamOptions?); // Promise - Call agent method +client.send(data); // void - Send raw WebSocket message +client.close(); // void - Close connection +client.reconnect(); // void - Force reconnection + +// Event listeners (inherited from PartySocket) +client.addEventListener("open", () => {}); +client.addEventListener("close", () => {}); +client.addEventListener("error", () => {}); +client.addEventListener("message", () => {}); +``` + +## MCP Server Integration + +If your agent uses MCP (Model Context Protocol) servers, you can receive updates about their state: + +```typescript +const agent = useAgent({ + agent: "AssistantAgent", + name: "session-123", + onMcpUpdate: (mcpServers) => { + // mcpServers is a record of server states + for (const [serverId, server] of Object.entries(mcpServers)) { + console.log(`${serverId}: ${server.connectionState}`); + console.log(`Tools: ${server.tools?.map((t) => t.name).join(", ")}`); + } + } +}); +``` + +## Error Handling + +### Connection Errors + +```typescript +const agent = useAgent({ + agent: "MyAgent", + onError: (error) => { + console.error("WebSocket error:", error); + }, + onClose: () => { + console.log("Connection closed, will auto-reconnect..."); + } +}); +``` + +### RPC Errors + +```typescript +try { + const result = await agent.call("riskyMethod", [data]); +} catch (error) { + // Error thrown by the agent method + console.error("RPC failed:", error.message); +} +``` + +### Streaming Errors + +```typescript +await agent.call("streamingMethod", [data], { + onChunk: (chunk) => handleChunk(chunk), + onError: (errorMessage) => { + // Stream-specific error handling + console.error("Stream error:", errorMessage); + } +}); +``` + +## Best Practices + +### 1. Use Typed Stubs + +```typescript +// Prefer this: +const user = await agent.stub.getUser(id); + +// Over this: +const user = await agent.call("getUser", [id]); +``` + +### 2. Reconnection is Automatic + +The client auto-reconnects and the agent automatically sends the current state on each connection. `agent.state` is updated automatically — no manual re-sync needed. + +### 3. Optimize Query Caching + +```typescript +// For auth tokens that expire hourly: +useAgent({ + query: async () => ({ token: await getToken() }), + cacheTtl: 55 * 60 * 1000, // Refresh 5 min before expiry + queryDeps: [userId] // Refresh if user changes +}); +``` + +### 4. Clean Up Connections + +In vanilla JS, close connections when done: + +```typescript +const client = new AgentClient({ agent: "MyAgent", host: "..." }); + +// When done: +client.close(); +``` + +React's `useAgent` handles cleanup automatically on unmount. diff --git a/docs/client-tools-continuation.md b/docs/client-tools-continuation.md new file mode 100644 index 0000000000..89fb81bafa --- /dev/null +++ b/docs/client-tools-continuation.md @@ -0,0 +1,177 @@ +# Client-Side Tools and Auto-Continuation + +## Overview + +Tools in `AIChatAgent` can be divided into two categories: + +- **Server tools**: Have an `execute` function on the server. The AI SDK runs them automatically and the LLM continues responding in the same turn. +- **Client tools**: No `execute` function on the server. The tool call is sent to the client via `onToolCall`, and the client provides the result. By default, this requires a new request to continue. + +With `autoContinueAfterToolResult`, client tools can behave like server tools -- the LLM calls a tool, the client executes it, and the server automatically continues the conversation in the same turn. + +## Server Setup + +Define a tool without an `execute` function. The AI SDK will pause and send `tool-input-available` to the client: + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { createWorkersAI } from "workers-ai-provider"; +import { streamText, tool, convertToModelMessages, stepCountIs } from "ai"; +import { z } from "zod"; + +export class MyAgent extends AIChatAgent { + async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages), + tools: { + // Client-side tool: no execute function + getUserLocation: tool({ + description: "Get the user's location from their browser", + inputSchema: z.object({}) + }), + + // Server-side tool: has execute, runs automatically + getWeather: tool({ + description: "Get weather for a city", + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }) => fetchWeather(city) + }) + }, + stopWhen: stepCountIs(5) // Allow multi-step so the LLM can respond after tool results + }); + + return result.toUIMessageStreamResponse(); + } +} +``` + +## Client Setup + +Use `onToolCall` to handle client-side tool execution. Auto-continuation is enabled by default (`autoContinueAfterToolResult: true`), so the server automatically calls `onChatMessage()` again after receiving the tool result, letting the LLM continue in the same assistant message. + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "MyAgent" }); + + const { messages, sendMessage } = useAgentChat({ + agent, + // Auto-continuation is enabled by default — no need to set this explicitly + // autoContinueAfterToolResult: true, + onToolCall: async ({ toolCall, addToolOutput }) => { + if (toolCall.toolName === "getUserLocation") { + const pos = await new Promise((resolve, reject) => { + navigator.geolocation.getCurrentPosition(resolve, reject); + }); + addToolOutput({ + toolCallId: toolCall.toolCallId, + output: { + lat: pos.coords.latitude, + lng: pos.coords.longitude + } + }); + } + } + }); + + // Render messages... +} +``` + +## How It Works + +``` +User: "What's the weather near me?" + +1. Client sends message → Server calls LLM +2. LLM decides to call getUserLocation (no server execute) +3. Stream sends tool-input-available to client +4. onToolCall fires → client gets geolocation → sends CF_AGENT_TOOL_RESULT +5. Server receives result with autoContinue: true +6. Server waits for the original stream to complete +7. Server calls onChatMessage() again (continuation) +8. LLM sees the location result, calls getWeather (server execute) +9. LLM responds: "It's sunny and 72°F near you!" +10. Continuation parts are merged into the same assistant message +``` + +The user sees a single seamless response, even though it involved a client-side tool call mid-stream. + +## Without Auto-Continuation + +When `autoContinueAfterToolResult` is set to `false`, the client must explicitly send a follow-up message after providing the tool result: + +```tsx +const { messages, sendMessage, addToolOutput } = useAgentChat({ + agent, + onToolCall: async ({ toolCall, addToolOutput: provide }) => { + if (toolCall.toolName === "getUserLocation") { + const pos = await getPosition(); + provide({ + toolCallId: toolCall.toolCallId, + output: { lat: pos.coords.latitude, lng: pos.coords.longitude } + }); + } + } + autoContinueAfterToolResult: false, // Disable auto-continuation +}); + +// After tool result is provided, send a follow-up to continue +// This creates a new assistant message rather than continuing the existing one +``` + +Use this when you want explicit control over when the conversation continues, or when tool results need user review before proceeding. + +## Combining with `needsApproval` + +You can use client-side tools and approval together. For example, a tool that needs both user approval and browser execution: + +```typescript +// Server: tool with needsApproval but no execute +const shareLocation = tool({ + description: "Share the user's location with a third party", + inputSchema: z.object({ service: z.string() }), + needsApproval: true + // No execute - client handles after approval +}); +``` + +```tsx +// Client: handle approval, then execute +const { addToolApprovalResponse } = useAgentChat({ + agent, + autoContinueAfterToolResult: true, + onToolCall: async ({ toolCall, addToolOutput }) => { + if (toolCall.toolName === "shareLocation") { + const pos = await getPosition(); + addToolOutput({ + toolCallId: toolCall.toolCallId, + output: { lat: pos.coords.latitude, lng: pos.coords.longitude } + }); + } + } +}); +``` + +The flow becomes: LLM calls tool → user approves → client executes → server auto-continues. + +If the user denies the tool instead, you can provide a custom error message using `addToolOutput` with `state: "output-error"`: + +```tsx +// Deny with a reason instead of generic rejection +addToolOutput({ + toolCallId: toolCall.toolCallId, + state: "output-error", + errorText: "User declined to share location" +}); +``` + +## Related Docs + +- [Chat Agents](./chat-agents.md) — Full `AIChatAgent` and `useAgentChat` reference +- [Human in the Loop](./human-in-the-loop.md) — Approval patterns including `needsApproval` diff --git a/docs/codemode.md b/docs/codemode.md new file mode 100644 index 0000000000..88cc7ea335 --- /dev/null +++ b/docs/codemode.md @@ -0,0 +1,300 @@ +# Codemode (Experimental) + +Codemode lets LLMs write and execute code that orchestrates your tools, instead of calling them one at a time. Inspired by [CodeAct](https://machinelearning.apple.com/research/codeact), it works because LLMs are better at writing code than making individual tool calls — they have seen millions of lines of real-world TypeScript but only contrived tool-calling examples. + +The `@cloudflare/codemode` package converts your tools into typed TypeScript APIs, gives the LLM a single "write code" tool, and executes the generated code in a secure, isolated Worker sandbox. + +> **Experimental** — this feature may have breaking changes in future releases. Use with caution in production. + +## When to use Codemode + +Codemode is most useful when the LLM needs to: + +- **Chain multiple tool calls** with logic between them (conditionals, loops, error handling) +- **Compose results** from different tools before returning +- **Work with MCP servers** that expose many fine-grained operations +- **Perform multi-step workflows** that would require many round-trips with standard tool calling + +For simple, single tool calls, standard AI SDK tool calling is simpler and sufficient. + +## Installation + +```sh +npm install @cloudflare/codemode ai zod +``` + +## Quick start + +### 1. Define your tools + +Use the standard AI SDK `tool()` function: + +```typescript +import { tool } from "ai"; +import { z } from "zod"; + +const tools = { + getWeather: tool({ + description: "Get weather for a location", + inputSchema: z.object({ location: z.string() }), + execute: async ({ location }) => `Weather in ${location}: 72°F, sunny` + }), + sendEmail: tool({ + description: "Send an email", + inputSchema: z.object({ + to: z.string(), + subject: z.string(), + body: z.string() + }), + execute: async ({ to, subject, body }) => `Email sent to ${to}` + }) +}; +``` + +### 2. Create the codemode tool + +`createCodeTool` takes your tools and an executor, and returns a single AI SDK tool: + +```typescript +import { createCodeTool } from "@cloudflare/codemode/ai"; +import { DynamicWorkerExecutor } from "@cloudflare/codemode"; + +const executor = new DynamicWorkerExecutor({ + loader: env.LOADER +}); + +const codemode = createCodeTool({ tools, executor }); +``` + +### 3. Use it with streamText + +Pass the codemode tool to `streamText` or `generateText` like any other tool. You choose the model: + +```typescript +import { streamText } from "ai"; + +const result = streamText({ + model, + system: "You are a helpful assistant.", + messages, + tools: { codemode } +}); +``` + +When the LLM decides to use codemode, it writes an async arrow function like: + +```javascript +async () => { + const weather = await codemode.getWeather({ location: "London" }); + if (weather.includes("sunny")) { + await codemode.sendEmail({ + to: "team@example.com", + subject: "Nice day!", + body: `It's ${weather}` + }); + } + return { weather, notified: true }; +}; +``` + +The code runs in an isolated Worker sandbox, tool calls are dispatched back to the host via Workers RPC, and the result is returned to the LLM. + +## Configuration + +### Wrangler bindings + +Add a `worker_loaders` binding to your `wrangler.jsonc`. This is the only binding required: + +```jsonc +// wrangler.jsonc +{ + "worker_loaders": [{ "binding": "LOADER" }], + "compatibility_flags": ["nodejs_compat"] +} +``` + +### Vite configuration + +If you use `zod-to-ts` (which codemode depends on), add a `__filename` define to your Vite config: + +```typescript +// vite.config.ts +export default defineConfig({ + plugins: [react(), cloudflare(), tailwindcss()], + define: { + __filename: "'index.ts'" + } +}); +``` + +## How it works + +``` +┌─────────────┐ ┌──────────────────────────────────────┐ +│ │ │ Dynamic Worker (isolated sandbox) │ +│ Host │ RPC │ │ +│ Worker │◄──────►│ LLM-generated code runs here │ +│ │ │ codemode.myTool() → dispatcher.call()│ +│ ToolDispatcher │ │ +│ holds tool fns │ fetch() blocked by default │ +└─────────────┘ └──────────────────────────────────────┘ +``` + +1. `createCodeTool` generates TypeScript type definitions from your tools and builds a description the LLM can read +2. The LLM writes an async arrow function that calls `codemode.toolName(args)` +3. The code is normalized via AST parsing (acorn) and sent to the executor +4. `DynamicWorkerExecutor` spins up an isolated Worker via `WorkerLoader` +5. Inside the sandbox, a `Proxy` intercepts `codemode.*` calls and routes them back to the host via Workers RPC (`ToolDispatcher extends RpcTarget`) +6. Console output (`console.log`, `console.warn`, `console.error`) is captured and returned in the result + +### Network isolation + +External `fetch()` and `connect()` are **blocked by default** — enforced at the Workers runtime level via `globalOutbound: null`. Sandboxed code can only interact with the host through `codemode.*` tool calls. + +To allow controlled outbound access, pass a `Fetcher`: + +```typescript +const executor = new DynamicWorkerExecutor({ + loader: env.LOADER, + globalOutbound: null // default — fully isolated + // globalOutbound: env.MY_OUTBOUND_SERVICE // route through a Fetcher +}); +``` + +## Using with an Agent + +The typical pattern is to create the executor and codemode tool inside an Agent's message handler: + +```typescript +import { Agent } from "agents"; +import { createCodeTool } from "@cloudflare/codemode/ai"; +import { DynamicWorkerExecutor } from "@cloudflare/codemode"; +import { streamText, convertToModelMessages, stepCountIs } from "ai"; + +export class MyAgent extends Agent { + async onChatMessage() { + const executor = new DynamicWorkerExecutor({ + loader: this.env.LOADER + }); + + const codemode = createCodeTool({ + tools: myTools, + executor + }); + + const result = streamText({ + model, + system: "You are a helpful assistant.", + messages: await convertToModelMessages(this.state.messages), + tools: { codemode }, + stopWhen: stepCountIs(10) + }); + + // Stream response back to client... + } +} +``` + +### With MCP tools + +MCP tools work the same way — merge them into the tool set: + +```typescript +const codemode = createCodeTool({ + tools: { + ...myTools, + ...this.mcp.getAITools() + }, + executor +}); +``` + +Tool names with hyphens or dots (common in MCP) are automatically sanitized to valid JavaScript identifiers (e.g., `my-server.list-items` becomes `my_server_list_items`). + +## The Executor interface + +The `Executor` interface is deliberately minimal — implement it to run code in any sandbox: + +```typescript +interface Executor { + execute( + code: string, + fns: Record Promise> + ): Promise; +} + +interface ExecuteResult { + result: unknown; + error?: string; + logs?: string[]; +} +``` + +`DynamicWorkerExecutor` is the built-in Cloudflare Workers implementation. You can build your own for Node VM, QuickJS, containers, or any other sandbox. + +## API reference + +### `createCodeTool(options)` + +Returns an AI SDK compatible `Tool`. + +| Option | Type | Default | Description | +| ------------- | ---------------------------- | -------------- | ------------------------------------------------------ | +| `tools` | `ToolSet \| ToolDescriptors` | required | Your tools (AI SDK `tool()` or raw descriptors) | +| `executor` | `Executor` | required | Where to run the generated code | +| `description` | `string` | auto-generated | Custom tool description. Use `{{types}}` for type defs | + +### `DynamicWorkerExecutor` + +Executes code in an isolated Cloudflare Worker via `WorkerLoader`. + +| Option | Type | Default | Description | +| ---------------- | ----------------- | -------- | ------------------------------------------------------------ | +| `loader` | `WorkerLoader` | required | Worker Loader binding from `env.LOADER` | +| `timeout` | `number` | `30000` | Execution timeout in ms | +| `globalOutbound` | `Fetcher \| null` | `null` | Network access control. `null` = blocked, `Fetcher` = routed | + +### `generateTypes(tools)` + +Generates TypeScript type definitions from your tools. Used internally by `createCodeTool` but exported for custom use (e.g., displaying types in a frontend). + +```typescript +import { generateTypes } from "@cloudflare/codemode"; + +const types = generateTypes(myTools); +// Returns: +// type CreateProjectInput = { name: string; description?: string } +// declare const codemode: { createProject: (input: CreateProjectInput) => Promise; } +``` + +### `sanitizeToolName(name)` + +Converts tool names into valid JavaScript identifiers. + +```typescript +import { sanitizeToolName } from "@cloudflare/codemode"; + +sanitizeToolName("get-weather"); // "get_weather" +sanitizeToolName("3d-render"); // "_3d_render" +sanitizeToolName("delete"); // "delete_" +``` + +## Security considerations + +- Code runs in **isolated Worker sandboxes** — each execution gets its own Worker instance +- External network access (`fetch`, `connect`) is **blocked by default** at the runtime level +- Tool calls are dispatched via Workers RPC, not network requests +- Execution has a configurable **timeout** (default 30 seconds) +- Console output is captured separately and does not leak to the host + +## Current limitations + +- **Tool approval (`needsApproval`) is not supported yet.** Tools with `needsApproval: true` execute immediately inside the sandbox without pausing for approval. Support for approval flows within codemode is planned. For now, do not pass approval-required tools to `createCodeTool` — use them through standard AI SDK tool calling instead. +- Requires Cloudflare Workers environment for `DynamicWorkerExecutor` +- Limited to JavaScript execution +- The `zod-to-ts` dependency bundles the TypeScript compiler, which increases Worker size +- LLM code quality depends on prompt engineering and model capability + +## Example + +See [`examples/codemode/`](../examples/codemode/) for a full working example — a project management assistant that uses codemode to orchestrate tasks, sprints, and comments via SQLite. diff --git a/docs/configuration.md b/docs/configuration.md new file mode 100644 index 0000000000..741b811dce --- /dev/null +++ b/docs/configuration.md @@ -0,0 +1,775 @@ +# Configuration + +This guide covers everything you need to configure agents for local development and production deployment, including wrangler.jsonc setup, type generation, environment variables, and the Cloudflare dashboard. + +## wrangler.jsonc + +The `wrangler.jsonc` file configures your Cloudflare Worker and its bindings. Here's a complete example for an agents project: + +```jsonc +{ + "$schema": "node_modules/wrangler/config-schema.json", + "name": "my-agent-app", + "main": "src/server.ts", + "compatibility_date": "2025-01-01", + "compatibility_flags": ["nodejs_compat"], + + // Static assets (optional) + "assets": { + "directory": "public", + "binding": "ASSETS" + }, + + // Durable Object bindings for agents + "durable_objects": { + "bindings": [ + { + "name": "MyAgent", + "class_name": "MyAgent" + }, + { + "name": "ChatAgent", + "class_name": "ChatAgent" + } + ] + }, + + // Required: Enable SQLite storage for agents + "migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["MyAgent", "ChatAgent"] + } + ], + + // AI binding (optional, for Workers AI) + "ai": { + "binding": "AI" + } +} +``` + +### Key Fields + +#### compatibility_flags + +The `nodejs_compat` flag is **required** for agents: + +```jsonc +"compatibility_flags": ["nodejs_compat"] +``` + +This enables Node.js compatibility mode, which agents depend on for crypto, streams, and other Node.js APIs. + +#### durable_objects.bindings + +Each agent class needs a binding: + +```jsonc +"durable_objects": { + "bindings": [ + { + "name": "Counter", // Property name on `env` (env.Counter) + "class_name": "Counter" // Exported class name (must match exactly) + } + ] +} +``` + +| Field | Description | +| ------------ | ----------------------------------------------------------- | +| `name` | The property name on `env`. Use this in code: `env.Counter` | +| `class_name` | Must match the exported class name exactly | + +**When name and class_name differ:** + +```jsonc +{ + "name": "COUNTER_DO", // env.COUNTER_DO + "class_name": "CounterAgent" // export class CounterAgent +} +``` + +This is useful when you want environment variable-style naming (`COUNTER_DO`) but more descriptive class names (`CounterAgent`). + +#### migrations + +Migrations tell Cloudflare how to set up storage for your Durable Objects: + +```jsonc +"migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["MyAgent"] + } +] +``` + +| Field | Description | +| -------------------- | ------------------------------------------------------------- | +| `tag` | Version identifier (e.g., "v1", "v2"). Must be unique | +| `new_sqlite_classes` | Agent classes that use SQLite storage (state persistence) | +| `deleted_classes` | Classes being removed | +| `renamed_classes` | Classes being renamed (see [Migrations](#migrations-1) below) | + +#### assets + +For serving static files (HTML, CSS, JS): + +```jsonc +"assets": { + "directory": "public", // Folder containing static files + "binding": "ASSETS" // Optional: binding for programmatic access +} +``` + +With a binding, you can serve assets programmatically: + +```typescript +export default { + async fetch(request: Request, env: Env) { + // static assets are served by the worker automatically by default + + // route the request to the appropriate agent + const agentResponse = await routeAgentRequest(request, env); + if (agentResponse) return agentResponse; + + // add your own routing logic here if you want to handle requests that are not for agents + return new Response("Not found", { status: 404 }); + } +}; +``` + +#### ai + +For Workers AI integration: + +```jsonc +"ai": { + "binding": "AI", + "remote": true // Mandatory: use remote inference (for local dev) +} +``` + +Access in your agent: + +```typescript +const response = await this.env.AI.run("@cf/moonshotai/kimi-k2.5", { + prompt: "Hello!" +}); +``` + +## TypeScript Configuration + +The Agents SDK ships a shared `tsconfig.json` that sets all the compiler options needed for agents projects — including the `ES2021` target required for `@callable()` decorators, strict mode, bundler module resolution, and Workers types. + +Extend it in your `tsconfig.json`: + +```json +{ + "extends": "agents/tsconfig" +} +``` + +This is equivalent to: + +```json +{ + "compilerOptions": { + "target": "ES2021", + "lib": ["ES2022", "DOM", "DOM.Iterable"], + "jsx": "react-jsx", + "module": "ES2022", + "moduleResolution": "bundler", + "types": ["node", "@cloudflare/workers-types", "vite/client"], + "allowImportingTsExtensions": true, + "noEmit": true, + "isolatedModules": true, + "verbatimModuleSyntax": true, + "esModuleInterop": true, + "forceConsistentCasingInFileNames": true, + "strict": true, + "skipLibCheck": true + } +} +``` + +You can override individual options as needed: + +```json +{ + "extends": "agents/tsconfig", + "compilerOptions": { + "jsx": "preserve" + } +} +``` + +> **Warning:** Do not set `"experimentalDecorators": true`. The Agents SDK uses [TC39 standard decorators](https://github.com/tc39/proposal-decorators), not TypeScript legacy decorators. Enabling `experimentalDecorators` applies an incompatible transform that silently breaks `@callable()` at runtime. + +## Vite Configuration + +The Agents SDK provides a Vite plugin that handles TC39 decorator transforms. Vite 8 uses Oxc for transpilation, which does not yet support TC39 decorators — without this plugin, `@callable()` and other decorators will fail at runtime. + +Add the plugin to your `vite.config.ts`: + +```typescript +import { cloudflare } from "@cloudflare/vite-plugin"; +import react from "@vitejs/plugin-react"; +import agents from "agents/vite"; +import { defineConfig } from "vite"; + +export default defineConfig({ + plugins: [agents(), react(), cloudflare()] +}); +``` + +The `agents()` plugin is safe to include even if your project does not use decorators. It only runs the transform on files that contain `@` syntax. + +The starter template and all examples include this plugin by default. + +## Generating Types + +Wrangler can generate TypeScript types for your bindings. + +### Automatic Generation + +Run the types command: + +```bash +npx wrangler types +``` + +This creates or updates `worker-configuration.d.ts` with your `Env` type. + +### Custom Output Path + +Specify a custom path: + +```bash +npx wrangler types env.d.ts +``` + +### Without Runtime Types + +For cleaner output (recommended for agents): + +```bash +npx wrangler types env.d.ts --include-runtime false +``` + +This generates just your bindings without Cloudflare runtime types. + +### Example Generated Output + +```typescript +// env.d.ts (generated) +declare namespace Cloudflare { + interface Env { + OPENAI_API_KEY: string; + Counter: DurableObjectNamespace; + ChatAgent: DurableObjectNamespace; + } +} +interface Env extends Cloudflare.Env {} +``` + +### Manual Type Definition + +You can also define types manually: + +```typescript +// env.d.ts +import type { Counter } from "./src/agents/counter"; +import type { ChatAgent } from "./src/agents/chat"; + +interface Env { + // Secrets + OPENAI_API_KEY: string; + WEBHOOK_SECRET: string; + + // Agent bindings + Counter: DurableObjectNamespace; + ChatAgent: DurableObjectNamespace; + + // Other bindings + AI: Ai; + ASSETS: Fetcher; + MY_KV: KVNamespace; +} +``` + +### Adding to package.json + +Add a script for easy regeneration: + +```json +{ + "scripts": { + "types": "wrangler types env.d.ts --include-runtime false" + } +} +``` + +## Environment Variables & Secrets + +### Local Development (.env) + +Create a `.env` file for local secrets (add to `.gitignore`): + +```bash +# .env +OPENAI_API_KEY=sk-... +GITHUB_WEBHOOK_SECRET=whsec_... +DATABASE_URL=postgres://... +``` + +Access in your agent: + +```typescript +class MyAgent extends Agent { + async onStart() { + const apiKey = process.env.OPENAI_API_KEY; + } +} +``` + +### Production Secrets + +Use `wrangler secret` for production: + +```bash +# Add a secret +wrangler secret put OPENAI_API_KEY +# Enter value when prompted + +# List secrets +wrangler secret list + +# Delete a secret +wrangler secret delete OPENAI_API_KEY +``` + +### Non-Secret Variables + +For non-sensitive configuration, use `vars` in wrangler.jsonc: + +```jsonc +{ + "vars": { + "API_BASE_URL": "https://api.example.com", + "MAX_RETRIES": "3", + "DEBUG_MODE": "false" + } +} +``` + +Note: All values must be strings. Parse numbers/booleans in code: + +```typescript +const maxRetries = parseInt(process.env.MAX_RETRIES, 10); +const debugMode = process.env.DEBUG_MODE === "true"; +``` + +### Environment-Specific Variables + +Use `[env.{name}]` sections for different environments (e.g. staging, production): + +```jsonc +{ + "name": "my-agent", + "vars": { + "API_URL": "https://api.example.com" + }, + + "env": { + "staging": { + "vars": { + "API_URL": "https://staging-api.example.com" + } + }, + "production": { + "vars": { + "API_URL": "https://api.example.com" + } + } + } +} +``` + +Deploy to specific environment: + +```bash +wrangler deploy --env staging +wrangler deploy --env production +``` + +## Local Development + +### Starting the Dev Server + +With Vite (recommended for full stack apps): + +```bash +npx vite dev +``` + +Without Vite: + +```bash +npx wrangler dev +``` + +### Local State Persistence + +Durable Object state is persisted locally in `.wrangler/state/`: + +``` +.wrangler/ +└── state/ + └── v3/ + └── d1/ + └── miniflare-D1DatabaseObject/ + └── ... (SQLite files) +``` + +### Clearing Local State + +To reset all local Durable Object state: + +```bash +rm -rf .wrangler/state +``` + +Or restart with fresh state: + +```bash +npx wrangler dev --persist-to="" +``` + +### Inspecting Local SQLite + +You can inspect agent state directly: + +```bash +# Find the SQLite file +ls .wrangler/state/v3/d1/ + +# Open with sqlite3 +sqlite3 .wrangler/state/v3/d1/miniflare-D1DatabaseObject/*.sqlite +``` + +## Dashboard Setup + +### Automatic Resources + +When you deploy, Cloudflare automatically creates: + +- **Worker** - Your deployed code +- **Durable Object namespaces** - One per agent class +- **SQLite storage** - Attached to each namespace + +### Viewing Durable Objects + +1. Go to [dash.cloudflare.com](https://dash.cloudflare.com) +2. Select your account → Workers & Pages +3. Click your Worker +4. Go to **Durable Objects** tab + +Here you can: + +- See all Durable Object namespaces +- View individual object instances +- Inspect storage (keys and values) +- Delete objects + +### Real-time Logs + +View live logs from your agents: + +```bash +npx wrangler tail +``` + +Or in the dashboard: + +1. Go to your Worker +2. Click **Logs** tab +3. Enable real-time logs + +Filter by: + +- Status (success, error) +- Search text +- Sampling rate + +### Analytics + +The dashboard shows: + +- Request count +- Error rate +- CPU time +- Duration percentiles +- Durable Object metrics + +## Production Deployment + +### Basic Deploy + +```bash +npx wrangler deploy +``` + +This: + +1. Bundles your code +2. Uploads to Cloudflare +3. Applies migrations +4. Makes it live on `*.workers.dev` + +### Custom Domain + +Add a route in wrangler.jsonc: + +```jsonc +{ + "routes": [ + { + "pattern": "agents.example.com/*", + "zone_name": "example.com" + } + ] +} +``` + +Or use a custom domain (simpler): + +```jsonc +{ + "routes": [ + { + "pattern": "agents.example.com", + "custom_domain": true + } + ] +} +``` + +### Preview Deployments + +Deploy without affecting production: + +```bash +npx wrangler deploy --dry-run # See what would be uploaded +npx wrangler versions upload # Upload new version +npx wrangler versions deploy # Gradually roll out +``` + +### Rollbacks + +Roll back to a previous version: + +```bash +npx wrangler rollback +``` + +## Multi-Environment Setup + +### Environment Configuration + +Define environments in wrangler.jsonc: + +```jsonc +{ + "name": "my-agent", + "main": "src/server.ts", + + // Base configuration (shared) + "compatibility_date": "2025-01-01", + "compatibility_flags": ["nodejs_compat"], + "durable_objects": { + "bindings": [{ "name": "MyAgent", "class_name": "MyAgent" }] + }, + "migrations": [{ "tag": "v1", "new_sqlite_classes": ["MyAgent"] }], + + // Environment overrides + "env": { + "staging": { + "name": "my-agent-staging", + "vars": { + "ENVIRONMENT": "staging" + } + }, + "production": { + "name": "my-agent-production", + "vars": { + "ENVIRONMENT": "production" + } + } + } +} +``` + +### Deploying to Environments + +```bash +# Deploy to staging +npx wrangler deploy --env staging + +# Deploy to production +npx wrangler deploy --env production + +# Set secrets per environment +npx wrangler secret put OPENAI_API_KEY --env staging +npx wrangler secret put OPENAI_API_KEY --env production +``` + +### Separate Durable Objects + +Each environment gets its own Durable Objects. Staging agents don't share state with production agents. + +To explicitly separate: + +```jsonc +{ + "env": { + "staging": { + "durable_objects": { + "bindings": [ + { + "name": "MyAgent", + "class_name": "MyAgent", + "script_name": "my-agent-staging" // Different namespace + } + ] + } + } + } +} +``` + +## Migrations + +Migrations manage Durable Object storage schema changes. + +### Adding a New Agent + +Add to `new_sqlite_classes` in a new migration: + +```jsonc +"migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["ExistingAgent"] + }, + { + "tag": "v2", + "new_sqlite_classes": ["NewAgent"] + } +] +``` + +### Renaming an Agent Class + +Use `renamed_classes`: + +```jsonc +"migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["OldName"] + }, + { + "tag": "v2", + "renamed_classes": [ + { + "from": "OldName", + "to": "NewName" + } + ] + } +] +``` + +**Important:** Also update: + +1. The class name in code +2. The `class_name` in bindings +3. Export statements + +### Deleting an Agent Class + +Use `deleted_classes`: + +```jsonc +"migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["AgentToDelete", "AgentToKeep"] + }, + { + "tag": "v2", + "deleted_classes": ["AgentToDelete"] + } +] +``` + +**Warning:** This permanently deletes all data for that class. + +### Migration Best Practices + +1. **Never modify existing migrations** - Always add new ones +2. **Use sequential tags** - v1, v2, v3 (or use dates: 2025-01-15) +3. **Test locally first** - Migrations run on deploy +4. **Back up production data** - Before renaming or deleting + +## Troubleshooting + +### "No such Durable Object class" + +The class isn't in migrations: + +```jsonc +"migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["MissingClassName"] // Add it here + } +] +``` + +### "Cannot find module" in types + +Regenerate types: + +```bash +npx wrangler types env.d.ts --include-runtime false +``` + +### Secrets not loading locally + +Check that `.env` exists and contains the variable: + +```bash +cat .env +# Should show: MY_SECRET=value +``` + +### Migration tag conflict + +Migration tags must be unique. If you see conflicts: + +```jsonc +// Wrong - duplicate tags +"migrations": [ + { "tag": "v1", "new_sqlite_classes": ["A"] }, + { "tag": "v1", "new_sqlite_classes": ["B"] } // Error! +] + +// Correct - sequential tags +"migrations": [ + { "tag": "v1", "new_sqlite_classes": ["A"] }, + { "tag": "v2", "new_sqlite_classes": ["B"] } +] +``` diff --git a/docs/cross-domain-authentication.md b/docs/cross-domain-authentication.md new file mode 100644 index 0000000000..530152050c --- /dev/null +++ b/docs/cross-domain-authentication.md @@ -0,0 +1,171 @@ +# Cross-Domain Authentication + +When your Agents are deployed, to keep things secure, send a token from the client, then verify it on the server. This mirrors the shape used in PartyKit’s auth guide. + +## WebSocket authentication + +WebSockets are not HTTP, so the handshake is limited when making cross-domain connections. + +**What you cannot send** + +- Custom headers during the upgrade +- `Authorization: Bearer ...` on connect + +**What works** + +- Put a signed, short-lived token in the connection URL as query parameters +- Verify the token in your server’s connect path + +> Tip: never place raw secrets in URLs. Prefer a JWT or a signed token that expires quickly and is scoped to the user or room. + +### Same origin + +If the client and server share the origin, the browser will send cookies during the WebSocket handshake. Session based auth can work here. Prefer HTTP-only cookies. + +### Cross origin + +Cookies do not help across origins. Pass credentials in the URL query, then verify on the server. + +## Usage examples + +### Static authentication + +```ts +import { useAgent } from "agents/react"; + +function ChatComponent() { + const agent = useAgent({ + agent: "my-agent", + query: { + token: "demo-token-123", + userId: "demo-user" + } + }); + + // Use agent to make calls, access state, etc. +} +``` + +### Async authentication + +Build query values right before connect. Use Suspense for async setup. + +```ts +import { useAgent } from "agents/react"; +import { Suspense, useCallback } from "react"; + +function ChatComponent() { + const asyncQuery = useCallback(async () => { + const [token, user] = await Promise.all([getAuthToken(), getCurrentUser()]); + return { + token, + userId: user.id, + timestamp: Date.now().toString() + }; + }, []); + + const agent = useAgent({ + agent: "my-agent", + query: asyncQuery + }); + + // Use agent to make calls, access state, etc. +} + +Authenticating...}> + + +``` + +### JWT refresh pattern + +Refresh the token when the connection fails due to authentication error. + +```ts +import { useAgent } from "agents/react"; +import { useCallback, useEffect } from "react"; + +const validateToken = async (token: string) => { + // An example of how you might implement this + const res = await fetch(`${API_HOST}/api/users/me`, { + headers: { + Authorization: `Bearer ${token}` + } + }); + + return res.ok; +}; + +const refreshToken = () => { + // Depends on implementation: + // - You could use a longer-lived token to refresh the expired token + // - De-auth the app and prompt the user to log in manually + // - ... +}; + +function useJWTAgent(agentName: string) { + const asyncQuery = useCallback(async () => { + let token = localStorage.getItem("jwt"); + + // If no token OR the token is no longer valid + // request a fresh token + if (!token && !(await validateToken(token))) { + token = await refreshToken(); + localStorage.setItem("jwt", token); + } + + return { + token + }; + }, []); + + const agent = useAgent({ + agent: agentName, + query: asyncQuery, + queryDeps: [] // Run on mount + }); +} +``` + +## Cross-domain authentication + +Pass credentials in the URL when connecting to another host, then verify on the server. + +```ts +import { useAgent } from "agents/react"; +import { useCallback } from "react"; + +// Static cross-domain auth +function StaticCrossDomainAuth() { + const agent = useAgent({ + agent: "my-agent", + host: "http://localhost:8788", + query: { + token: "demo-token-123", + userId: "demo-user" + } + }); + + // Use agent to make calls, access state, etc. +} + +// Async cross-domain auth +function AsyncCrossDomainAuth() { + const asyncQuery = useCallback(async () => { + const [token, user] = await Promise.all([getAuthToken(), getCurrentUser()]); + return { + token, + userId: user.id, + timestamp: Date.now().toString() + }; + }, []); + + const agent = useAgent({ + agent: "my-agent", + host: "http://localhost:8788", + query: asyncQuery + }); + + // Use agent to make calls, access state, etc. +} +``` diff --git a/docs/durable-execution.md b/docs/durable-execution.md new file mode 100644 index 0000000000..9c280d8d4f --- /dev/null +++ b/docs/durable-execution.md @@ -0,0 +1,342 @@ +# Durable Execution + +Run work that survives Durable Object eviction. `runFiber()` registers a task in SQLite, keeps the agent alive during execution, lets you checkpoint intermediate state with `stash()`, and calls `onFiberRecovered()` on the next activation if the agent was evicted mid-task. + +> For how fibers fit into the bigger picture of building agents that run for weeks or months, see [Long-Running Agents](./long-running-agents.md). + +## Quick start + +```typescript +import { Agent } from "agents"; +import type { FiberRecoveryContext } from "agents"; + +class MyAgent extends Agent { + async doWork() { + await this.runFiber("my-task", async (ctx) => { + const step1 = await expensiveOperation(); + ctx.stash({ step1 }); + + const step2 = await anotherExpensiveOperation(step1); + this.setState({ ...this.state, result: step2 }); + }); + } + + async onFiberRecovered(ctx: FiberRecoveryContext) { + if (ctx.name !== "my-task") return; + const snapshot = ctx.snapshot as { step1: unknown } | null; + if (snapshot) { + // Resume from the checkpoint — step1 is done, run step2 + const step2 = await anotherExpensiveOperation(snapshot.step1); + this.setState({ ...this.state, result: step2 }); + } + } +} +``` + +## Why fibers exist + +Durable Objects get evicted for three reasons: + +1. **Inactivity timeout** — ~70–140 seconds with no incoming requests or open WebSockets +2. **Code updates / runtime restarts** — non-deterministic, 1–2x per day +3. **Alarm handler timeout** — 15 minutes + +When eviction happens mid-work, the upstream HTTP connection (to an LLM provider, an API, a database) is severed permanently. In-memory state — streaming buffers, partial responses, loop counters — is lost. Multi-turn agent loops lose their position entirely. + +`keepAlive()` reduces the chance of eviction. `runFiber()` makes eviction survivable. + +For work that should run independently of the agent with per-step retries and multi-step orchestration, use [Workflows](./workflows.md) instead. Fibers are for work that is part of the agent's own execution. See [Long-Running Agents: Workflows vs agent-internal patterns](./long-running-agents.md#when-to-use-workflows-vs-agent-internal-patterns) for a comparison. + +## `keepAlive()` + +Prevents idle eviction by creating a 30-second alarm heartbeat that resets the inactivity timer. + +```typescript +class Agent { + keepAlive(): Promise<() => void>; + keepAliveWhile(fn: () => Promise): Promise; +} +``` + +`keepAliveWhile()` is the recommended approach — it runs an async function and automatically cleans up the heartbeat when it completes or throws: + +```typescript +const result = await this.keepAliveWhile(async () => { + return await slowAPICall(); +}); +``` + +For manual control, `keepAlive()` returns a disposer. Always call it when done — otherwise the heartbeat continues indefinitely: + +```typescript +const dispose = await this.keepAlive(); +try { + await longWork(); +} finally { + dispose(); +} +``` + +### How it works + +While any `keepAlive` ref is held, an alarm fires every 30 seconds that resets the inactivity timer. When all disposers are called, alarms stop and the DO can go idle naturally. + +The heartbeat is invisible to `getSchedules()` — no schedule rows are created. It does not conflict with your own schedules; the alarm system multiplexes all schedules and the keepAlive heartbeat through a single alarm slot. + +### Configurable interval + +Default: 30 seconds. The inactivity timeout is ~70–140 seconds, so 30 seconds gives comfortable margin. Override via static options: + +```typescript +class MyAgent extends Agent { + static options = { keepAliveIntervalMs: 2_000 }; // useful for testing recovery locally +} +``` + +### When to use keepAlive vs runFiber + +`keepAlive` prevents eviction but does nothing about recovery. If the agent _is_ evicted despite the heartbeat (code update, alarm timeout, resource limit), any in-progress work is lost. + +`runFiber` calls `keepAlive` internally _and_ persists the work in SQLite so it can be recovered. Use `keepAlive` alone when the work is cheap to redo or does not need checkpointing. Use `runFiber` when the work is expensive and you need to resume from where you left off. + +| Scenario | Use | +| ------------------------------------------------ | --------------------------- | +| Waiting on a slow API call | `keepAlive()` | +| Streaming an LLM response (via `AIChatAgent`) | Automatic (built in) | +| Multi-step computation with intermediate results | `runFiber()` | +| Background research loop that takes 10+ minutes | `runFiber()` with `stash()` | + +## `runFiber()` + +Durable execution with checkpointing and recovery. + +```typescript +class Agent { + runFiber(name: string, fn: (ctx: FiberContext) => Promise): Promise; + stash(data: unknown): void; + onFiberRecovered(ctx: FiberRecoveryContext): Promise; +} + +type FiberContext = { + id: string; + stash(data: unknown): void; + snapshot: unknown | null; +}; + +type FiberRecoveryContext = { + id: string; + name: string; + snapshot: unknown | null; +}; +``` + +### Lifecycle + +#### Normal execution + +``` +runFiber("work", fn) + ├─ INSERT row into cf_agents_runs + ├─ keepAlive() — heartbeat starts + ├─ Execute fn(ctx) + │ ├─ ctx.stash(data) → UPDATE snapshot in SQLite + │ ├─ ctx.stash(data) → UPDATE snapshot in SQLite + │ └─ return result + ├─ DELETE row from cf_agents_runs + ├─ keepAlive dispose — heartbeat stops + └─ Return result to caller +``` + +#### Eviction and recovery + +``` +[DO evicted — all in-memory state lost] + + On next activation: + ├─ Request/connection → onStart() → _checkRunFibers() [primary path] + │ OR + ├─ Persisted alarm fires → _onAlarmHousekeeping() [fallback path] + + _checkRunFibers(): + ├─ SELECT * FROM cf_agents_runs + ├─ For each orphaned row: + │ ├─ Parse snapshot from JSON + │ ├─ Call onFiberRecovered(ctx) + │ └─ DELETE the row + └─ If onFiberRecovered calls runFiber() again → new row, normal execution +``` + +Both recovery paths call the same hook. The alarm path is critical for background agents that have no incoming client connections — the persisted alarm wakes the agent on its own. + +#### Error during execution + +``` +fn(ctx) throws Error + ├─ DELETE row from cf_agents_runs + ├─ keepAlive dispose + └─ Error propagates to caller (or logged if fire-and-forget) +``` + +No automatic retries. Recovery logic belongs in `onFiberRecovered`, where you have the snapshot and full context about what went wrong. + +### Inline vs fire-and-forget + +`runFiber()` supports both patterns: + +```typescript +// Inline — await the result +const result = await this.runFiber("work", async (ctx) => { + return computeExpensiveThing(); +}); + +// Fire-and-forget — caller does not wait +void this.runFiber("background", async (ctx) => { + await longRunningProcess(); +}); +``` + +If the DO is evicted during an inline `await`, the caller is gone. On recovery, `onFiberRecovered` fires — it cannot return a result to the original caller. This is the inherent limitation of durable execution across process boundaries. For long-running work that is likely to outlive a single DO lifetime, fire-and-forget with checkpoint/recovery is the safer pattern. + +## Checkpoints with `stash()` + +`ctx.stash(data)` writes to SQLite **synchronously**. There is no async gap between "I decided to save" and "it is saved." If eviction happens after `stash()` returns, the data is guaranteed to be in SQLite. + +Each call **fully replaces** the previous snapshot — it is not a merge. Write the complete recovery state you need: + +```typescript +await this.runFiber("research", async (ctx) => { + const steps = ["search", "analyze", "synthesize"]; + const completed: string[] = []; + const results: Record = {}; + + for (const step of steps) { + results[step] = await executeStep(step); + completed.push(step); + + ctx.stash({ + completed, + results, + pendingSteps: steps.slice(completed.length) + }); + } +}); +``` + +### `this.stash()` vs `ctx.stash()` + +Both do the same thing. `ctx.stash()` uses a direct closure over the fiber ID. `this.stash()` uses `AsyncLocalStorage` to find the currently executing fiber — it works correctly even with concurrent fibers, since each fiber's ALS context is independent. + +`this.stash()` is convenient when calling from nested functions that do not have access to `ctx`. It throws if called outside a `runFiber` callback. + +## Recovery + +Override `onFiberRecovered` to handle interrupted fibers. The default implementation logs a warning and deletes the row. + +```typescript +class ResearchAgent extends Agent { + async onFiberRecovered(ctx: FiberRecoveryContext) { + if (ctx.name !== "research") return; + + const snapshot = ctx.snapshot as { + completed: string[]; + results: Record; + pendingSteps: string[]; + } | null; + + if (snapshot && snapshot.pendingSteps.length > 0) { + // Resume from where we left off + void this.runFiber("research", async (fiberCtx) => { + const { completed, results, pendingSteps } = snapshot; + + for (const step of pendingSteps) { + results[step] = await this.executeStep(step); + completed.push(step); + + fiberCtx.stash({ + completed, + results, + pendingSteps: pendingSteps.slice(pendingSteps.indexOf(step) + 1) + }); + } + }); + } + } +} +``` + +Key points: + +- **The original lambda is gone.** On recovery, you only have the `name` and `snapshot`. The lambda cannot be serialized — recovery logic must be in the hook. +- **The row is deleted after the hook runs.** If you want to continue the work, call `runFiber()` again inside the hook — this creates a new row. +- **You control what recovery means.** Retry from the beginning, resume from a checkpoint, skip and notify the user, or do nothing. The framework does not impose a strategy. +- **If the hook throws, the row is still deleted.** You do not get a second chance at recovery. If your recovery logic can fail, catch errors and handle them (e.g., schedule a retry, log, or re-create the fiber). + +### Chat recovery + +`AIChatAgent` builds on fibers for LLM streaming recovery. When `chatRecovery` is enabled, each chat turn is wrapped in a fiber automatically. The framework handles the internal recovery path and exposes `onChatRecovery` for provider-specific strategies. See [Long-Running Agents: Recovering interrupted LLM streams](./long-running-agents.md#recovering-interrupted-llm-streams) and the [`forever-chat` example](../experimental/forever-chat/). + +## Concurrent fibers + +Multiple fibers can run at the same time. Each has its own row in SQLite with its own snapshot, and each calls `keepAlive()` independently (ref-counted, so the DO stays alive until all fibers complete). + +```typescript +// Run two fibers concurrently +void this.runFiber("fetch-data", async (ctx) => { + /* ... */ +}); +void this.runFiber("process-queue", async (ctx) => { + /* ... */ +}); +``` + +On recovery, `_checkRunFibers()` iterates all orphaned rows and calls `onFiberRecovered` for each. Use `ctx.name` to distinguish between fiber types in your recovery hook. + +## Testing locally + +In `wrangler dev`, fiber recovery works identically to production. SQLite and alarm state persist to disk between restarts. + +1. Start your agent and trigger a fiber (`runFiber`) +2. Kill the wrangler process (Ctrl-C or SIGKILL) +3. Restart wrangler +4. Recovery fires automatically — via `onStart()` if a request arrives, or via the persisted alarm if no clients connect + +The E2E test in `packages/agents/src/e2e-tests/` validates this: it starts wrangler, spawns a fiber, kills the process with SIGKILL, restarts with the same persist directory, and verifies the fiber recovers automatically. + +## API reference + +### `runFiber(name, fn)` + +Execute a durable fiber. The fiber is registered in SQLite before `fn` runs and deleted after it completes (or throws). `keepAlive()` is held for the duration. + +- **`name`** — identifier for the fiber, used in `onFiberRecovered` to distinguish fiber types. Not unique — multiple fibers can share a name. +- **`fn`** — async function receiving a `FiberContext`. Closures work naturally (`this` and local variables are captured). +- **Returns** — the value returned by `fn`. If the DO is evicted before completion, the return value is lost; recovery happens through the hook. + +### `stash(data)` / `ctx.stash(data)` + +Checkpoint the current fiber's state. Writes synchronously to SQLite. Each call fully replaces the previous snapshot. `data` must be JSON-serializable. + +### `onFiberRecovered(ctx)` + +Called once per orphaned fiber row on agent restart. Override to implement recovery. The row is deleted after this hook returns. + +- **`ctx.id`** — unique fiber ID +- **`ctx.name`** — the name passed to `runFiber()` +- **`ctx.snapshot`** — the last `stash()` data, or `null` if `stash()` was never called + +### `keepAlive()` + +Create a 30-second alarm heartbeat. Returns a disposer function. Idempotent — calling the disposer multiple times is safe. + +### `keepAliveWhile(fn)` + +Run an async function while keeping the DO alive. Heartbeat starts before `fn` and stops when it completes or throws. Returns the value returned by `fn`. + +## Related + +- [Long-Running Agents](./long-running-agents.md) — how fibers compose with schedules, plans, and async operations +- [Scheduling](./scheduling.md) — `keepAlive` details and the alarm system +- [Workflows](./workflows.md) — durable multi-step execution outside the agent +- [Chat Agents](./chat-agents.md) — `chatRecovery` and `onChatRecovery` +- [`forever-chat` example](../experimental/forever-chat/) — multi-provider LLM recovery demo +- [`forever.md` design doc](../experimental/forever.md) — internal design details, tradeoffs, and architecture diff --git a/docs/email.md b/docs/email.md new file mode 100644 index 0000000000..aed6e95cfc --- /dev/null +++ b/docs/email.md @@ -0,0 +1,449 @@ +# Email Routing + +Agents can receive and process emails using Cloudflare's [Email Routing](https://developers.cloudflare.com/email-routing/email-workers/). This guide covers how to route inbound emails to your Agents and handle replies securely. + +## Prerequisites + +1. A domain configured with [Cloudflare Email Routing](https://developers.cloudflare.com/email-routing/) +2. An Email Worker configured to receive emails +3. An Agent to process emails + +## Quick Start + +```ts +import { Agent, routeAgentEmail } from "agents"; +import { createAddressBasedEmailResolver, type AgentEmail } from "agents/email"; + +// Your Agent that handles emails +export class EmailAgent extends Agent { + async onEmail(email: AgentEmail) { + console.log("Received email from:", email.from); + console.log("Subject:", email.headers.get("subject")); + + // Reply to the email + await this.replyToEmail(email, { + fromName: "My Agent", + body: "Thanks for your email!" + }); + } +} + +// Route emails to your Agent +export default { + async email(message, env) { + await routeAgentEmail(message, env, { + resolver: createAddressBasedEmailResolver("EmailAgent") + }); + } +}; +``` + +## Resolvers + +Resolvers determine which Agent instance receives an incoming email. Choose the resolver that matches your use case. + +### createAddressBasedEmailResolver + +**Recommended for inbound mail.** Routes emails based on the recipient address. + +```ts +import { createAddressBasedEmailResolver } from "agents/email"; + +const resolver = createAddressBasedEmailResolver("EmailAgent"); +``` + +**Routing logic:** + +| Recipient Address | Agent Name | Agent ID | +| --------------------------------------- | ---------------------- | --------- | +| `support@example.com` | `EmailAgent` (default) | `support` | +| `sales@example.com` | `EmailAgent` (default) | `sales` | +| `NotificationAgent+user123@example.com` | `NotificationAgent` | `user123` | + +The sub-address format (`agent+id@domain`) allows routing to different agent namespaces and instances from a single email domain. + +> **Note:** Agent class names in the recipient address are matched case-insensitively. Email infrastructure often lowercases addresses, so `NotificationAgent+user123@example.com` and `notificationagent+user123@example.com` both route to the `NotificationAgent` class. + +### createSecureReplyEmailResolver + +**For reply flows with signature verification.** Verifies that incoming emails are authentic replies to your outbound emails, preventing attackers from routing emails to arbitrary agent instances. + +```ts +import { createSecureReplyEmailResolver } from "agents/email"; + +const resolver = createSecureReplyEmailResolver(env.EMAIL_SECRET); +``` + +When your agent sends an email with `replyToEmail()` and a `secret`, it signs the routing headers with a timestamp. When a reply comes back, this resolver verifies the signature and checks that it hasn't expired before routing. + +**Options:** + +```ts +const resolver = createSecureReplyEmailResolver(env.EMAIL_SECRET, { + // Maximum age of signature in seconds (default: 30 days) + maxAge: 7 * 24 * 60 * 60, // 7 days + + // Callback for logging/debugging signature failures + onInvalidSignature: (email, reason) => { + console.warn(`Invalid signature from ${email.from}: ${reason}`); + // reason can be: "missing_headers", "expired", "invalid", "malformed_timestamp" + } +}); +``` + +**When to use:** If your agent initiates email conversations and you need replies to route back to the same agent instance securely. + +### createCatchAllEmailResolver + +**For single-instance routing.** Routes all emails to a specific agent instance regardless of the recipient address. + +```ts +import { createCatchAllEmailResolver } from "agents/email"; + +const resolver = createCatchAllEmailResolver("EmailAgent", "default"); +``` + +**When to use:** When you have a single agent instance that handles all emails (e.g., a shared inbox). + +### Combining Resolvers + +You can combine resolvers to handle different scenarios: + +```ts +export default { + async email(message, env) { + const secureReplyResolver = createSecureReplyEmailResolver( + env.EMAIL_SECRET + ); + const addressResolver = createAddressBasedEmailResolver("EmailAgent"); + + await routeAgentEmail(message, env, { + resolver: async (email, env) => { + // First, check if this is a signed reply + const replyRouting = await secureReplyResolver(email, env); + if (replyRouting) return replyRouting; + + // Otherwise, route based on recipient address + return addressResolver(email, env); + }, + + // Handle emails that don't match any routing rule + onNoRoute: (email) => { + console.warn(`No route found for email from ${email.from}`); + email.setReject("Unknown recipient"); + } + }); + } +}; +``` + +## Handling Emails in Your Agent + +### The AgentEmail Interface + +When your agent's `onEmail` method is called, it receives an `AgentEmail` object: + +```ts +type AgentEmail = { + from: string; // Sender's email address + to: string; // Recipient's email address + headers: Headers; // Email headers (subject, message-id, etc.) + rawSize: number; // Size of the raw email in bytes + + getRaw(): Promise; // Get the full raw email content + reply(options): Promise; // Send a reply + forward(rcptTo, headers?): Promise; // Forward the email + setReject(reason): void; // Reject the email with a reason +}; +``` + +### Parsing Email Content + +Use a library like [postal-mime](https://www.npmjs.com/package/postal-mime) to parse the raw email: + +```ts +import PostalMime from "postal-mime"; + +async onEmail(email: AgentEmail) { + const raw = await email.getRaw(); + const parsed = await PostalMime.parse(raw); + + console.log("Subject:", parsed.subject); + console.log("Text body:", parsed.text); + console.log("HTML body:", parsed.html); + console.log("Attachments:", parsed.attachments); +} +``` + +### Detecting Auto-Reply Emails + +Use `isAutoReplyEmail()` to detect auto-reply emails and avoid mail loops: + +```ts +import { isAutoReplyEmail } from "agents/email"; +import PostalMime from "postal-mime"; + +async onEmail(email: AgentEmail) { + const raw = await email.getRaw(); + const parsed = await PostalMime.parse(raw); + + // Detect auto-reply emails to avoid sending duplicate responses + if (isAutoReplyEmail(parsed.headers)) { + console.log("Skipping auto-reply email"); + return; + } + + // Process the email... +} +``` + +This checks for standard RFC 3834 headers (`Auto-Submitted`, `X-Auto-Response-Suppress`, `Precedence`) that indicate an email is an auto-reply. + +### Replying to Emails + +Use `this.replyToEmail()` to send a reply: + +```ts +async onEmail(email: AgentEmail) { + await this.replyToEmail(email, { + fromName: "Support Bot", // Display name for the sender + subject: "Re: Your inquiry", // Optional, defaults to "Re: " + body: "Thanks for contacting us!", // Email body + contentType: "text/plain", // Optional, defaults to "text/plain" + headers: { // Optional custom headers + "X-Custom-Header": "value" + }, + secret: this.env.EMAIL_SECRET // Optional, signs headers for secure reply routing + }); +} +``` + +### Forwarding Emails + +```ts +async onEmail(email: AgentEmail) { + await email.forward("admin@example.com"); +} +``` + +### Rejecting Emails + +```ts +async onEmail(email: AgentEmail) { + if (isSpam(email)) { + email.setReject("Message rejected as spam"); + return; + } + // Process the email... +} +``` + +## Secure Reply Routing + +When your agent sends emails and expects replies, use secure reply routing to prevent attackers from forging headers to route emails to arbitrary agent instances. + +### How It Works + +1. **Outbound:** When you call `replyToEmail()` with a `secret`, the agent signs the routing headers (`X-Agent-Name`, `X-Agent-ID`) using HMAC-SHA256 +2. **Inbound:** `createSecureReplyEmailResolver` verifies the signature before routing +3. **Enforcement:** If an email was routed via the secure resolver, `replyToEmail()` requires a secret (or explicit `null` to opt-out) + +### Setup + +1. Add a secret to your `wrangler.jsonc`: + +```jsonc +// wrangler.jsonc +{ + "vars": { + "EMAIL_SECRET": "change-me-in-production" + } +} +``` + +For production, use Wrangler secrets instead: + +```bash +wrangler secret put EMAIL_SECRET +``` + +2. Use the combined resolver pattern: + +```ts +export default { + async email(message, env) { + const secureReplyResolver = createSecureReplyEmailResolver( + env.EMAIL_SECRET + ); + const addressResolver = createAddressBasedEmailResolver("EmailAgent"); + + await routeAgentEmail(message, env, { + resolver: async (email, env) => { + const replyRouting = await secureReplyResolver(email, env); + if (replyRouting) return replyRouting; + return addressResolver(email, env); + } + }); + } +}; +``` + +3. Sign outbound emails: + +```ts +async onEmail(email: AgentEmail) { + await this.replyToEmail(email, { + fromName: "My Agent", + body: "Thanks for your email!", + secret: this.env.EMAIL_SECRET // Signs the routing headers + }); +} +``` + +### Enforcement Behavior + +When an email is routed via `createSecureReplyEmailResolver`, the `replyToEmail()` method enforces signing: + +| `secret` value | Behavior | +| --------------------- | ------------------------------------------------------------ | +| `"my-secret"` | Signs headers (secure) | +| `undefined` (omitted) | **Throws error** - must provide secret or explicit opt-out | +| `null` | Allowed but not recommended - explicitly opts out of signing | + +## Complete Example + +Here's a complete email agent with secure reply routing: + +```ts +import { Agent, routeAgentEmail } from "agents"; +import { + createAddressBasedEmailResolver, + createSecureReplyEmailResolver, + type AgentEmail +} from "agents/email"; +import PostalMime from "postal-mime"; + +interface Env { + EmailAgent: DurableObjectNamespace; + EMAIL_SECRET: string; +} + +export class EmailAgent extends Agent { + async onEmail(email: AgentEmail) { + const raw = await email.getRaw(); + const parsed = await PostalMime.parse(raw); + + console.log(`Email from ${email.from}: ${parsed.subject}`); + + // Store the email in state + const emails = this.state.emails || []; + emails.push({ + from: email.from, + subject: parsed.subject, + receivedAt: new Date().toISOString() + }); + this.setState({ ...this.state, emails }); + + // Send auto-reply with signed headers + await this.replyToEmail(email, { + fromName: "Support Bot", + body: `Thanks for your email! We received: "${parsed.subject}"`, + secret: this.env.EMAIL_SECRET + }); + } +} + +export default { + async email(message, env: Env) { + const secureReplyResolver = createSecureReplyEmailResolver( + env.EMAIL_SECRET, + { + maxAge: 7 * 24 * 60 * 60, // 7 days + onInvalidSignature: (email, reason) => { + console.warn(`Invalid signature from ${email.from}: ${reason}`); + } + } + ); + const addressResolver = createAddressBasedEmailResolver("EmailAgent"); + + await routeAgentEmail(message, env, { + resolver: async (email, env) => { + // Try secure reply routing first + const replyRouting = await secureReplyResolver(email, env); + if (replyRouting) return replyRouting; + // Fall back to address-based routing + return addressResolver(email, env); + }, + onNoRoute: (email) => { + console.warn(`No route found for email from ${email.from}`); + email.setReject("Unknown recipient"); + } + }); + } +} satisfies ExportedHandler; +``` + +## API Reference + +### routeAgentEmail + +```ts +function routeAgentEmail( + email: ForwardableEmailMessage, + env: Env, + options: { + resolver: EmailResolver; + onNoRoute?: (email: ForwardableEmailMessage) => void | Promise; + } +): Promise; +``` + +Routes an incoming email to the appropriate Agent based on the resolver's decision. + +| Option | Description | +| ----------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `resolver` | Function that determines which agent to route the email to | +| `onNoRoute` | Optional callback invoked when no routing information is found. Use this to reject the email or perform custom handling. If not provided, a warning is logged and the email is dropped. | + +### createSecureReplyEmailResolver + +```ts +function createSecureReplyEmailResolver( + secret: string, + options?: { + maxAge?: number; + onInvalidSignature?: ( + email: ForwardableEmailMessage, + reason: SignatureFailureReason + ) => void; + } +): EmailResolver; + +type SignatureFailureReason = + | "missing_headers" + | "expired" + | "invalid" + | "malformed_timestamp"; +``` + +Creates a resolver for routing email replies with signature verification. + +| Option | Description | +| -------------------- | ------------------------------------------------------------------------ | +| `secret` | Secret key for HMAC verification (must match the key used to sign) | +| `maxAge` | Maximum age of signature in seconds (default: 30 days / 2592000 seconds) | +| `onInvalidSignature` | Optional callback for logging when signature verification fails | + +### signAgentHeaders + +```ts +function signAgentHeaders( + secret: string, + agentName: string, + agentId: string +): Promise>; +``` + +Manually sign agent routing headers. Returns an object with `X-Agent-Name`, `X-Agent-ID`, `X-Agent-Sig`, and `X-Agent-Sig-Ts` headers. + +Useful when sending emails through external services while maintaining secure reply routing. The signature includes a timestamp and will be valid for 30 days by default. diff --git a/docs/get-current-agent.md b/docs/get-current-agent.md new file mode 100644 index 0000000000..961d8e1974 --- /dev/null +++ b/docs/get-current-agent.md @@ -0,0 +1,158 @@ +# getCurrentAgent() + +## Automatic Context for Custom Methods + +**All custom methods automatically have full agent context!** The framework automatically detects and wraps your custom methods during initialization, ensuring `getCurrentAgent()` works seamlessly everywhere. + +## How It Works + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { getCurrentAgent } from "agents"; + +export class MyAgent extends AIChatAgent { + async customMethod() { + const { agent } = getCurrentAgent(); + // ✅ agent is automatically available! + console.log(agent.name); + } + + async anotherMethod() { + // ✅ This works too - no setup needed! + const { agent } = getCurrentAgent(); + return agent.state; + } +} +``` + +**Zero configuration required!** The framework automatically: + +1. Scans your agent class for custom methods +2. Wraps them with agent context during initialization +3. Ensures `getCurrentAgent()` works in all external functions called from your methods + +## Real-World Example + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { getCurrentAgent } from "agents"; +import { generateText } from "ai"; +import { openai } from "@ai-sdk/openai"; + +// External utility function that needs agent context +async function processWithAI(prompt: string) { + const { agent } = getCurrentAgent(); + // ✅ External functions can access the current agent! + + return await generateText({ + model: openai("gpt-4"), + prompt: `Agent ${agent?.name}: ${prompt}` + }); +} + +export class MyAgent extends AIChatAgent { + async customMethod(message: string) { + // Use this.* to access agent properties directly + console.log("Agent name:", this.name); + console.log("Agent state:", this.state); + + // External functions automatically work! + const result = await processWithAI(message); + return result.text; + } +} +``` + +### Built-in vs Custom Methods + +- **Built-in methods** (onRequest, onEmail, onStateChanged): Already have context +- **Custom methods** (your methods): Automatically wrapped during initialization +- **External functions**: Access context through `getCurrentAgent()` + +### The Context Flow + +```typescript +// When you call a custom method: +agent.customMethod() + → automatically wrapped with agentContext.run() + → your method executes with full context + → external functions can use getCurrentAgent() +``` + +## Common Use Cases + +### Working with AI SDK Tools + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { generateText } from "ai"; +import { openai } from "@ai-sdk/openai"; + +export class MyAgent extends AIChatAgent { + async generateResponse(prompt: string) { + // AI SDK tools automatically work + const response = await generateText({ + model: openai("gpt-4"), + prompt, + tools: { + // Tools that use getCurrentAgent() work perfectly + } + }); + + return response.text; + } +} +``` + +### Calling External Libraries + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { getCurrentAgent } from "agents"; + +async function saveToDatabase(data: any) { + const { agent } = getCurrentAgent(); + // Can access agent info for logging, context, etc. + console.log(`Saving data for agent: ${agent?.name}`); +} + +export class MyAgent extends AIChatAgent { + async processData(data: any) { + // External functions automatically have context + await saveToDatabase(data); + } +} +``` + +## API Reference + +The agents package exports one main function for context management: + +### `getCurrentAgent()` + +Gets the current agent from any context where it's available. + +**Returns:** + +```typescript +{ + agent: T | undefined, + connection: Connection | undefined, + request: Request | undefined +} +``` + +**Usage:** + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { getCurrentAgent } from "agents"; + +export class MyAgent extends AIChatAgent { + async customMethod() { + const { agent, connection, request } = getCurrentAgent(); + // agent is properly typed as MyAgent + // connection and request available if called from a request handler + } +} +``` diff --git a/docs/getting-started.md b/docs/getting-started.md new file mode 100644 index 0000000000..d633101290 --- /dev/null +++ b/docs/getting-started.md @@ -0,0 +1,305 @@ +# Getting Started + +Build AI agents that persist, think, and act. Agents run on Cloudflare's global network, maintain state across requests, and connect to clients in real-time via WebSockets. + +**What you'll build:** A counter agent with persistent state that syncs to a React frontend in real-time. + +**Time:** ~10 minutes + +--- + +## Create a New Project + +```bash +npm create cloudflare@latest -- --template cloudflare/agents-starter +cd my-agent +npm install +``` + +This creates a project with: + +- `src/server.ts` - Your agent code +- `src/client.tsx` - React frontend +- `wrangler.jsonc` - Cloudflare configuration +- `tsconfig.json` - Extends `agents/tsconfig` for correct decorator and module settings +- `vite.config.ts` - Includes the `agents/vite` plugin for decorator support + +The starter template includes two SDK integrations that are required for `@callable()` decorators. If you are setting up a project manually, add both: + +**tsconfig.json** — extends `agents/tsconfig`, which sets `target: "ES2021"` and other recommended options: + +```json +{ + "extends": "agents/tsconfig" +} +``` + +**vite.config.ts** — includes the `agents()` plugin, which handles TC39 decorator transforms (required because Vite 8's Oxc transpiler does not support them yet): + +```typescript +import { cloudflare } from "@cloudflare/vite-plugin"; +import react from "@vitejs/plugin-react"; +import agents from "agents/vite"; +import { defineConfig } from "vite"; + +export default defineConfig({ + plugins: [agents(), react(), cloudflare()] +}); +``` + +Start the dev server: + +```bash +npm run dev +``` + +Open [http://localhost:5173](http://localhost:5173) to see your agent in action. + +--- + +## Your First Agent + +Let's build a simple counter agent from scratch. Replace `src/server.ts`: + +```typescript +import { Agent, routeAgentRequest, callable } from "agents"; + +// Define the state shape +type CounterState = { + count: number; +}; + +// Create the agent +export class Counter extends Agent { + // Initial state for new instances + initialState: CounterState = { count: 0 }; + + // Methods marked with @callable can be called from the client + @callable() + increment() { + this.setState({ count: this.state.count + 1 }); + return this.state.count; + } + + @callable() + decrement() { + this.setState({ count: this.state.count - 1 }); + return this.state.count; + } + + @callable() + reset() { + this.setState({ count: 0 }); + } +} + +// Route requests to agents +export default { + async fetch(request: Request, env: Env, ctx: ExecutionContext) { + return ( + (await routeAgentRequest(request, env)) ?? + new Response("Not found", { status: 404 }) + ); + } +}; +``` + +Update `wrangler.jsonc` to register the agent: + +```jsonc +{ + "name": "my-agent", + "main": "src/server.ts", + "compatibility_date": "2025-01-01", + "compatibility_flags": ["nodejs_compat"], + "durable_objects": { + "bindings": [ + { + "name": "Counter", + "class_name": "Counter" + } + ] + }, + "migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["Counter"] + } + ] +} +``` + +--- + +## Connect from React + +Replace `src/client.tsx`: + +```tsx +import { useAgent } from "agents/react"; + +// Match your agent's state type +type CounterState = { + count: number; +}; + +export default function App() { + // Connect to the Counter agent + const agent = useAgent({ + agent: "Counter" + }); + + return ( +
+

Counter Agent

+

{agent.state?.count ?? 0}

+
+ + + +
+
+ ); +} +``` + +Key points: + +- **`useAgent`** connects to your agent via WebSocket +- **`agent.state`** is reactive — the component re-renders when state changes +- **`agent.stub.methodName()`** calls methods marked with `@callable()` on your agent + +--- + +## What Just Happened? + +When you clicked the button: + +1. **Client** called `agent.stub.increment()` over WebSocket +2. **Agent** ran `increment()`, updated state with `setState()` +3. **State** persisted to SQLite automatically +4. **Broadcast** sent to all connected clients +5. **React** re-rendered with updated `agent.state` + +``` +┌─────────────┐ ┌─────────────┐ +│ Browser │◄───────►│ Agent │ +│ (React) │ WS │ (Counter) │ +└─────────────┘ └──────┬──────┘ + │ + ┌──────▼──────┐ + │ SQLite │ + │ (State) │ + └─────────────┘ +``` + +### Key Concepts + +| Concept | What it means | +| -------------------- | ------------------------------------------------------------------------------------------- | +| **Agent instance** | Each unique name gets its own agent. `Counter:user-123` is separate from `Counter:user-456` | +| **Persistent state** | State survives restarts, deploys, and hibernation. It's stored in SQLite | +| **Real-time sync** | All clients connected to the same agent receive state updates instantly | +| **Hibernation** | When no clients are connected, the agent hibernates (no cost). It wakes on the next request | + +--- + +## Connect from Vanilla JS + +If you're not using React: + +```typescript +import { AgentClient } from "agents/client"; + +const agent = new AgentClient({ + agent: "Counter", + name: "my-counter", // optional, defaults to "default" + host: window.location.host +}); + +await agent.ready; + +// Call methods +await agent.call("increment"); +console.log("Current count:", agent.state?.count); + +await agent.call("reset"); +``` + +--- + +## Deploy to Cloudflare + +```bash +npm run deploy +``` + +Your agent is now live on Cloudflare's global network, running close to your users. + +--- + +## Next Steps + +Now that you have a working agent, explore these topics: + +- **[State Management](./state.md)** - Deep dive into `setState()`, `initialState`, and `onStateChanged()` +- **[Client SDK](./client-sdk.md)** - Full `useAgent` and `AgentClient` API reference +- **[Scheduling](./scheduling.md)** - Run tasks on a delay, schedule, or cron +- **[Agent Class](./agent-class.md)** - Lifecycle methods, HTTP handlers, and WebSocket events + +### Common Patterns + +| I want to... | Read... | +| ------------------------ | ---------------------------------------- | +| Add AI/LLM capabilities | [Chat Agents](./chat-agents.md) | +| Expose tools via MCP | [Creating MCP Servers](./mcp-servers.md) | +| Run background tasks | [Scheduling](./scheduling.md) | +| Handle emails | [Email Routing](./email.md) | +| Use Cloudflare Workflows | [Workflows](./workflows.md) | + +--- + +## Troubleshooting + +### "Agent not found" / 404 errors + +Make sure: + +1. Agent class is exported from your server file +2. `wrangler.jsonc` has the binding and migration +3. Agent name in client matches the class name (case-insensitive) + +### State not syncing + +Check that: + +1. You're calling `this.setState()`, not mutating `this.state` directly +2. Your agent has `initialState` defined (state is only sent on connect if the agent has state) +3. WebSocket connection is established (check browser dev tools) + +### "Method X is not callable" errors + +Make sure your methods are decorated with `@callable()`: + +```typescript +import { callable } from "agents"; + +@callable() +increment() { + // ... +} +``` + +### Type errors with `agent.stub` + +Add the agent type parameter: + +```typescript +const agent = useAgent({ + agent: "Counter", + onStateUpdate: (state) => setCount(state.count) +}); + +// Now agent.stub is fully typed +agent.stub.increment(); // ✓ TypeScript knows this method exists +``` diff --git a/docs/http-websockets.md b/docs/http-websockets.md new file mode 100644 index 0000000000..b74e306df8 --- /dev/null +++ b/docs/http-websockets.md @@ -0,0 +1,668 @@ +# HTTP & WebSockets + +Agents handle both HTTP requests and WebSocket connections, giving you flexibility to build REST APIs, real-time applications, or hybrid architectures. + +## Overview + +Every agent can respond to: + +- **HTTP requests** via `onRequest()` - REST APIs, webhooks, file uploads +- **WebSocket connections** via `onConnect()`, `onMessage()`, `onClose()` - Real-time bidirectional communication + +```typescript +import { Agent } from "agents"; + +export class MyAgent extends Agent { + // Handle HTTP requests + onRequest(request: Request): Response { + return new Response("Hello from HTTP!"); + } + + // Handle WebSocket connections + onConnect(connection: Connection, ctx: ConnectionContext) { + connection.send("Welcome!"); + } + + onMessage(connection: Connection, message: WSMessage) { + // Echo back + connection.send(message); + } +} +``` + +## Lifecycle Hooks + +Agents have several lifecycle hooks that are called at different points: + +| Hook | When Called | +| --------------------------------------------- | --------------------------------------------------------- | +| `onStart(props?)` | Once when the agent first starts (before any connections) | +| `onRequest(request)` | When an HTTP request is received (non-WebSocket) | +| `onConnect(connection, ctx)` | When a new WebSocket connection is established | +| `onMessage(connection, message)` | When a WebSocket message is received | +| `onClose(connection, code, reason, wasClean)` | When a WebSocket connection closes | +| `onError(connection, error)` | When a WebSocket error occurs | +| `onError(error)` | When a server error occurs (overloaded) | + +### Lifecycle Flow + +``` +Agent Created + ↓ + onStart() ←── Called once, before any connections + ↓ +┌─────────────────────────────────────┐ +│ For each request: │ +│ │ +│ HTTP Request ──→ onRequest() │ +│ │ +│ WebSocket ──→ onConnect() │ +│ ↓ │ +│ Messages ──→ onMessage() (repeat) │ +│ ↓ │ +│ Disconnect ──→ onClose() │ +└─────────────────────────────────────┘ +``` + +## HTTP Requests + +Handle HTTP requests with `onRequest()`. This is called for any non-WebSocket request to your agent. + +```typescript +export class ApiAgent extends Agent { + async onRequest(request: Request): Promise { + const url = new URL(request.url); + + // Route by path + if (url.pathname.endsWith("/status")) { + return Response.json({ + status: "ok", + connections: this.getConnections().length + }); + } + + if (url.pathname.endsWith("/data") && request.method === "POST") { + const data = await request.json(); + // Process data... + return Response.json({ received: true }); + } + + return new Response("Not found", { status: 404 }); + } +} +``` + +### Common HTTP Patterns + +**REST API with CORS:** + +```typescript +onRequest(request: Request): Response { + // Handle preflight + if (request.method === "OPTIONS") { + return new Response(null, { + headers: { + "Access-Control-Allow-Origin": "*", + "Access-Control-Allow-Methods": "GET, POST, OPTIONS", + "Access-Control-Allow-Headers": "Content-Type" + } + }); + } + + return Response.json( + { data: "..." }, + { headers: { "Access-Control-Allow-Origin": "*" } } + ); +} +``` + +**File Upload:** + +```typescript +async onRequest(request: Request): Promise { + if (request.method === "POST") { + const formData = await request.formData(); + const file = formData.get("file") as File; + + // Process file... + const content = await file.text(); + + return Response.json({ filename: file.name, size: file.size }); + } + return new Response("Method not allowed", { status: 405 }); +} +``` + +## WebSocket Connections + +WebSockets enable real-time bidirectional communication between clients and your agent. + +### Connection Lifecycle + +```typescript +export class ChatAgent extends Agent { + // Called when a client connects + onConnect(connection: Connection, ctx: ConnectionContext) { + // ctx.request contains the original HTTP request (for auth, headers, etc.) + const url = new URL(ctx.request.url); + const token = url.searchParams.get("token"); + + console.log(`Client ${connection.id} connected`); + connection.send(JSON.stringify({ type: "welcome", id: connection.id })); + } + + // Called for each message from this connection + onMessage(connection: Connection, message: WSMessage) { + if (typeof message === "string") { + const data = JSON.parse(message); + // Handle message... + } else { + // Binary message (ArrayBuffer) + } + } + + // Called when connection closes + onClose( + connection: Connection, + code: number, + reason: string, + wasClean: boolean + ) { + console.log(`Client ${connection.id} disconnected: ${code} ${reason}`); + } + + // Called on WebSocket errors + onError(connection: Connection, error: unknown) { + console.error(`Error on connection ${connection.id}:`, error); + } +} +``` + +### The Connection Object + +Each WebSocket connection is represented by a `Connection` object: + +```typescript +interface Connection { + /** Unique connection identifier */ + id: string; + + /** The agent instance name this connection belongs to */ + server: string; + + /** Per-connection state (read-only, use setState to update) */ + state: TState | null; + + /** Update connection state */ + setState(state: TState | ((prev: TState | null) => TState)): void; + + /** Send a message to this connection */ + send(message: string | ArrayBuffer): void; + + /** Close this connection */ + close(code?: number, reason?: string): void; +} +``` + +### Message Types + +Messages can be strings or binary: + +```typescript +onMessage(connection: Connection, message: WSMessage) { + if (typeof message === "string") { + // Text message - usually JSON + const data = JSON.parse(message); + this.handleTextMessage(connection, data); + } else { + // Binary message - ArrayBuffer or ArrayBufferView + this.handleBinaryMessage(connection, message); + } +} +``` + +## Connection Management + +### Getting Connections + +```typescript +// Get all connections +const connections = this.getConnections(); + +// Get a specific connection by ID +const connection = this.getConnection("abc123"); + +// Get connections with a specific tag +const adminConnections = this.getConnections("admin"); +``` + +### Broadcasting + +Send a message to all connected clients: + +```typescript +// Broadcast to everyone +this.broadcast(JSON.stringify({ type: "update", data: "..." })); + +// Broadcast to everyone except specific connections +this.broadcast( + JSON.stringify({ type: "user-typing", userId: "123" }), + ["connection-id-to-exclude"] // Don't send to the originator +); +``` + +### Connection Tags + +Tag connections for easy filtering. Override `getConnectionTags()` to assign tags: + +```typescript +export class ChatAgent extends Agent { + // Called when a connection is established + getConnectionTags(connection: Connection, ctx: ConnectionContext): string[] { + const url = new URL(ctx.request.url); + const role = url.searchParams.get("role"); + + const tags: string[] = []; + if (role === "admin") tags.push("admin"); + if (role === "moderator") tags.push("moderator"); + + return tags; // Up to 9 tags, max 256 chars each + } + + // Later, broadcast only to admins + notifyAdmins(message: string) { + for (const conn of this.getConnections("admin")) { + conn.send(message); + } + } +} +``` + +## Per-Connection State + +Store data specific to each connection using `connection.state` and `connection.setState()`: + +```typescript +type ConnectionState = { + username: string; + joinedAt: number; + messageCount: number; +}; + +export class ChatAgent extends Agent { + onConnect(connection: Connection, ctx: ConnectionContext) { + const url = new URL(ctx.request.url); + + // Initialize connection state + connection.setState({ + username: url.searchParams.get("username") || "Anonymous", + joinedAt: Date.now(), + messageCount: 0 + }); + } + + onMessage(connection: Connection, message: WSMessage) { + // Update message count using functional update + connection.setState((prev) => ({ + ...prev!, + messageCount: (prev?.messageCount || 0) + 1 + })); + + // Access current state + const { username, messageCount } = connection.state!; + console.log(`${username} sent message #${messageCount}`); + } +} +``` + +**Important:** Connection state is: + +- **Immutable** - Read via `connection.state`, update via `connection.setState()` +- **Per-connection** - Each connection has its own state +- **Persisted across hibernation** - Survives agent sleep/wake cycles + +## The `onStart` Hook + +`onStart()` is called once when the agent first starts, before any connections are established: + +```typescript +export class MyAgent extends Agent { + private cache: Map = new Map(); + + async onStart() { + // Initialize resources + console.log(`Agent ${this.name} starting...`); + + // Load data from storage + const savedData = this.sql`SELECT * FROM cache`; + for (const row of savedData) { + this.cache.set(row.key, row.value); + } + + // Restore MCP connections, check workflows, etc. + // (Agent does this automatically, but you can add custom logic) + } + + onConnect(connection: Connection) { + // By the time connections arrive, onStart has completed + } +} +``` + +## Protocol Message Control + +By default, when a WebSocket client connects, the agent sends protocol text frames (`CF_AGENT_IDENTITY`, `CF_AGENT_STATE`, `CF_AGENT_MCP_SERVERS`) to keep the client in sync. You can suppress these on a per-connection basis by overriding `shouldSendProtocolMessages`: + +```typescript +export class MyAgent extends Agent { + shouldSendProtocolMessages( + connection: Connection, + ctx: ConnectionContext + ): boolean { + // Suppress protocol frames for binary-only clients + const url = new URL(ctx.request.url); + return url.searchParams.get("protocol") !== "false"; + } +} +``` + +When `shouldSendProtocolMessages` returns `false` for a connection: + +- No `CF_AGENT_IDENTITY`, `CF_AGENT_STATE`, or `CF_AGENT_MCP_SERVERS` frames are sent on connect +- The connection is excluded from protocol broadcasts (state updates, MCP server changes) +- Regular messages via `connection.send()` and `this.broadcast()` still work normally + +This is useful for IoT devices, binary-only clients, or lightweight consumers that only need raw messages. + +### Checking Protocol Status + +Use `isConnectionProtocolEnabled` to check whether a connection receives protocol messages: + +```typescript +const enabled = this.isConnectionProtocolEnabled(connection); +``` + +This status persists across hibernation — a connection that was marked as no-protocol before hibernation remains no-protocol after waking up. + +## Error Handling + +Handle errors gracefully with `onError`: + +```typescript +export class MyAgent extends Agent { + // WebSocket connection error + onError(connection: Connection, error: unknown): void { + console.error(`WebSocket error on ${connection.id}:`, error); + connection.send( + JSON.stringify({ type: "error", message: "Connection error" }) + ); + } + + // Server error (overloaded signature - no connection parameter) + onError(error: unknown): void { + console.error("Server error:", error); + // Log to external service, etc. + } +} +``` + +## Hibernation + +Agents support hibernation - they can sleep when inactive and wake when needed. This saves resources while maintaining WebSocket connections. + +### Enabling Hibernation + +Hibernation is enabled by default. To disable: + +```typescript +export class AlwaysOnAgent extends Agent { + static options = { hibernate: false }; +} +``` + +### How Hibernation Works + +1. Agent is active, handling connections +2. After ~10 seconds of no messages, agent hibernates (sleeps) +3. WebSocket connections remain open (handled by Cloudflare) +4. When a message arrives, agent wakes up +5. `onMessage` is called as normal + +### What Persists Across Hibernation + +| Persists | Does Not Persist | +| -------------------------- | ------------------- | +| `this.state` (agent state) | In-memory variables | +| `connection.state` | Timers/intervals | +| SQLite data (`this.sql`) | Promises in flight | +| Connection metadata | Local caches | + +**Best Practice:** Store important data in `this.state` or SQLite, not in class properties: + +```typescript +export class MyAgent extends Agent { + initialState = { counter: 0 }; + + // ❌ Don't do this - lost on hibernation + private localCounter = 0; + + onMessage(connection: Connection, message: WSMessage) { + // ✅ Do this - persists + this.setState({ counter: this.state.counter + 1 }); + + // ❌ Lost after hibernation + this.localCounter++; + } +} +``` + +## Common Patterns + +### Authentication on Connect + +Validate users when they connect: + +```typescript +export class SecureAgent extends Agent { + async onConnect(connection: Connection, ctx: ConnectionContext) { + const url = new URL(ctx.request.url); + const token = url.searchParams.get("token"); + + if (!token || !(await this.validateToken(token))) { + connection.close(4001, "Unauthorized"); + return; + } + + const user = await this.getUserFromToken(token); + connection.setState({ userId: user.id, role: user.role }); + + connection.send(JSON.stringify({ type: "authenticated", user })); + } + + private async validateToken(token: string): Promise { + // Validate JWT, check database, etc. + return true; + } +} +``` + +### Chat Room with Broadcast + +```typescript +type Message = { + type: "message" | "join" | "leave"; + user: string; + text?: string; + timestamp: number; +}; + +export class ChatRoom extends Agent { + onConnect(connection: Connection, ctx: ConnectionContext) { + const url = new URL(ctx.request.url); + const username = url.searchParams.get("username") || "Anonymous"; + + connection.setState({ username }); + + // Notify others + this.broadcast( + JSON.stringify({ + type: "join", + user: username, + timestamp: Date.now() + } satisfies Message), + [connection.id] // Don't send to the joining user + ); + } + + onMessage(connection: Connection, message: WSMessage) { + if (typeof message !== "string") return; + + const { username } = connection.state as { username: string }; + + // Broadcast to everyone + this.broadcast( + JSON.stringify({ + type: "message", + user: username, + text: message, + timestamp: Date.now() + } satisfies Message) + ); + } + + onClose(connection: Connection) { + const { username } = (connection.state as { username: string }) || {}; + if (username) { + this.broadcast( + JSON.stringify({ + type: "leave", + user: username, + timestamp: Date.now() + } satisfies Message) + ); + } + } +} +``` + +### Presence Tracking + +Track who's online using per-connection state. This pattern is clean because connection state is automatically cleaned up when users disconnect: + +```typescript +type UserState = { + name: string; + joinedAt: number; + lastSeen: number; +}; + +export class PresenceAgent extends Agent { + onConnect(connection: Connection, ctx: ConnectionContext) { + const url = new URL(ctx.request.url); + const name = url.searchParams.get("name") || "Anonymous"; + + // Store user data on the connection itself + connection.setState({ + name, + joinedAt: Date.now(), + lastSeen: Date.now() + }); + + // Send current presence to new user + connection.send( + JSON.stringify({ + type: "presence", + users: this.getPresence() + }) + ); + + // Notify others that someone joined + this.broadcastPresence(); + } + + onClose(connection: Connection) { + // No manual cleanup needed - connection state is automatically gone + // Just broadcast updated presence to remaining users + this.broadcastPresence(); + } + + // Heartbeat to update lastSeen + onMessage(connection: Connection, message: WSMessage) { + if (message === "ping") { + connection.setState((prev) => ({ + ...prev!, + lastSeen: Date.now() + })); + connection.send("pong"); + } + } + + // Build presence from all connections + private getPresence() { + const users: Record = {}; + for (const conn of this.getConnections()) { + if (conn.state) { + users[conn.id] = { + name: conn.state.name, + lastSeen: conn.state.lastSeen + }; + } + } + return users; + } + + private broadcastPresence() { + this.broadcast( + JSON.stringify({ + type: "presence", + users: this.getPresence() + }) + ); + } +} +``` + +## API Reference + +### Agent Lifecycle Methods + +| Method | Signature | Description | +| ---------------------------- | --------------------------------------------------------------- | ------------------------------------------------------ | +| `onStart` | `(props?) => void \| Promise` | Called once when agent starts | +| `onRequest` | `(request: Request) => Response \| Promise` | Handle HTTP requests | +| `onConnect` | `(connection, ctx) => void \| Promise` | WebSocket connected | +| `onMessage` | `(connection, message) => void \| Promise` | Message received | +| `onClose` | `(connection, code, reason, wasClean) => void \| Promise` | Connection closed | +| `onError` | `(connection, error) => void \| Promise` | WebSocket error | +| `onError` | `(error) => void \| Promise` | Server error (overload) | +| `shouldSendProtocolMessages` | `(connection, ctx) => boolean` | Control per-connection protocol frames (default: true) | + +### Connection Management Methods + +| Method | Signature | Description | +| ----------------------------- | ----------------------------------------- | ---------------------------------------------- | +| `getConnections` | `(tag?: string) => Iterable` | Get all connections, optionally by tag | +| `getConnection` | `(id: string) => Connection \| undefined` | Get connection by ID | +| `getConnectionTags` | `(connection, ctx) => string[]` | Override to tag connections | +| `broadcast` | `(message, without?: string[]) => void` | Send to all connections | +| `isConnectionProtocolEnabled` | `(connection) => boolean` | Check if connection receives protocol messages | + +### Connection Object + +| Property/Method | Type | Description | +| --------------- | ------------------------------------------ | -------------------------------- | +| `id` | `string` | Unique connection identifier | +| `server` | `string` | Agent instance name | +| `state` | `T \| null` | Per-connection state (read-only) | +| `setState` | `(state \| (prev) => state) => void` | Update connection state | +| `send` | `(message: string \| ArrayBuffer) => void` | Send message | +| `close` | `(code?, reason?) => void` | Close connection | + +### Agent Properties + +| Property | Type | Description | +| ------------ | -------------------- | ----------------------------------- | +| `this.name` | `string` | Agent instance name | +| `this.state` | `State` | Agent state (use with `setState()`) | +| `this.env` | `Env` | Environment bindings | +| `this.ctx` | `DurableObjectState` | Durable Object context | diff --git a/docs/human-in-the-loop.md b/docs/human-in-the-loop.md new file mode 100644 index 0000000000..3397bb6f37 --- /dev/null +++ b/docs/human-in-the-loop.md @@ -0,0 +1,655 @@ +# Human in the Loop + +Human-in-the-loop (HITL) patterns allow agents to pause execution and wait for human approval, confirmation, or input before proceeding. This is essential for compliance, safety, and oversight in agentic systems. + +## Overview + +### Why Human in the Loop? + +- **Compliance**: Regulatory requirements may mandate human approval for certain actions +- **Safety**: High-stakes operations (payments, deletions, external communications) need oversight +- **Quality**: Human review catches errors AI might miss +- **Trust**: Users feel more confident when they can approve critical actions + +### Common Use Cases + +| Use Case | Example | +| ------------------- | ---------------------------------------- | +| Financial approvals | Expense reports, payment processing | +| Content moderation | Publishing, email sending | +| Data operations | Bulk deletions, exports | +| AI tool execution | Confirming LLM tool calls before running | +| Access control | Granting permissions, role changes | + +## Choosing an Approach + +Agents SDK supports multiple human-in-the-loop patterns. Choose based on your use case: + +| Use Case | Pattern | Best For | Example | +| ---------------------- | ----------------- | -------------------------------------------------- | ----------------------------------------------------------------- | +| Long-running workflows | Workflow Approval | Multi-step processes, durable approval gates | [examples/workflows/](../examples/workflows/) | +| AIChatAgent tools | `needsApproval` | Chat-based tool calls with `@cloudflare/ai-chat` | [guides/human-in-the-loop/](../guides/human-in-the-loop/) | +| OpenAI Agents SDK | `needsApproval` | Using OpenAI's agent SDK with conditional approval | [openai-sdk/human-in-the-loop/](../openai-sdk/human-in-the-loop/) | +| Client-side tools | `onToolCall` | Tools that need browser APIs or user interaction | Pattern below | +| MCP Servers | Elicitation | MCP tools requesting structured user input | [examples/mcp-elicitation/](../examples/mcp-elicitation/) | + +### Decision Guide + +``` +Is this part of a multi-step workflow? +├── Yes → Use Workflow Approval (waitForApproval) +└── No → Are you building an MCP server? + ├── Yes → Use MCP Elicitation (elicitInput) + └── No → Is this an AI chat interaction? + ├── Yes → Does the tool need browser APIs? + │ ├── Yes → Use onToolCall (client-side execution) + │ └── No → Use needsApproval (server-side with approval) + └── No → Use State + WebSocket for simple confirmations +``` + +## Workflow-Based Approval + +For durable, multi-step processes, use Cloudflare Workflows with the `waitForApproval()` helper. The workflow pauses until a human approves or rejects. + +### Basic Pattern + +```typescript +import { Agent, AgentWorkflow, callable } from "agents"; +import type { AgentWorkflowEvent, AgentWorkflowStep } from "agents"; + +// Workflow that pauses for approval +export class ExpenseWorkflow extends AgentWorkflow< + ExpenseAgent, + ExpenseParams +> { + async run(event: AgentWorkflowEvent, step: AgentWorkflowStep) { + const expense = event.payload; + + // Step 1: Validate the expense + const validated = await step.do("validate", async () => { + return validateExpense(expense); + }); + + // Step 2: Wait for manager approval + await this.reportProgress({ + step: "approval", + status: "pending", + message: `Awaiting approval for $${expense.amount}` + }); + + // This pauses the workflow until approved/rejected + const approval = await this.waitForApproval<{ approvedBy: string }>(step, { + timeout: "7 days" + }); + + console.log(`Approved by: ${approval.approvedBy}`); + + // Step 3: Process the approved expense + const result = await step.do("process", async () => { + return processExpense(validated); + }); + + await step.reportComplete(result); + return result; + } +} +``` + +### Agent Methods for Approval + +The agent provides methods to approve or reject waiting workflows: + +```typescript +export class ExpenseAgent extends Agent { + initialState: ExpenseState = { + pendingApprovals: [], + status: "idle" + }; + + // Approve a waiting workflow + @callable() + async approve(workflowId: string, approvedBy: string): Promise { + await this.approveWorkflow(workflowId, { + reason: "Expense approved", + metadata: { approvedBy, approvedAt: Date.now() } + }); + + // Update state to reflect approval + this.setState({ + ...this.state, + pendingApprovals: this.state.pendingApprovals.filter( + (p) => p.workflowId !== workflowId + ) + }); + } + + // Reject a waiting workflow + @callable() + async reject(workflowId: string, reason: string): Promise { + await this.rejectWorkflow(workflowId, { reason }); + + this.setState({ + ...this.state, + pendingApprovals: this.state.pendingApprovals.filter( + (p) => p.workflowId !== workflowId + ) + }); + } + + // Track workflow progress + async onWorkflowProgress( + workflowName: string, + workflowId: string, + progress: unknown + ): Promise { + const p = progress as { step: string; status: string }; + + if (p.step === "approval" && p.status === "pending") { + // Add to pending approvals list + this.setState({ + ...this.state, + pendingApprovals: [ + ...this.state.pendingApprovals, + { workflowId, requestedAt: Date.now() } + ] + }); + } + } +} +``` + +### Timeout Handling + +Set timeouts to prevent workflows from waiting indefinitely: + +```typescript +const approval = await this.waitForApproval(step, { + timeout: "7 days" // or "1 hour", "30 minutes", etc. +}); +``` + +If the timeout expires, the workflow continues without approval data. Handle this case: + +```typescript +const approval = await this.waitForApproval<{ approvedBy: string }>(step, { + timeout: "24 hours" +}); + +if (!approval) { + // Timeout expired - escalate or auto-reject + await step.reportError("Approval timeout - escalating to manager"); + throw new Error("Approval timeout"); +} +``` + +For more details, see [Workflows Integration](./workflows.md). + +## AI Tool Approval with `needsApproval` + +When building AI chat agents, you often want humans to approve certain tool calls before execution. The AI SDK's `needsApproval` option pauses tool execution until the user approves or rejects. + +### Server + +Define tools with `needsApproval` to require human confirmation: + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { createWorkersAI } from "workers-ai-provider"; +import { streamText, tool, convertToModelMessages } from "ai"; +import { z } from "zod"; + +export class MyAgent extends AIChatAgent { + async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages), + tools: { + // Tool with conditional approval + processPayment: tool({ + description: "Process a payment", + inputSchema: z.object({ + amount: z.number(), + recipient: z.string() + }), + // Approval required for amounts over $100 + needsApproval: async ({ amount }) => amount > 100, + execute: async ({ amount, recipient }) => { + return await chargeCard(amount, recipient); + } + }), + + // Tool that always requires approval + deleteAccount: tool({ + description: "Delete a user account", + inputSchema: z.object({ userId: z.string() }), + needsApproval: true, + execute: async ({ userId }) => { + return await deleteUser(userId); + } + }), + + // Tool that executes automatically (no approval) + getWeather: tool({ + description: "Get weather for a city", + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }) => fetchWeather(city) + }) + }, + maxSteps: 5 + }); + + return result.toUIMessageStreamResponse(); + } +} +``` + +### Client + +Handle approval requests with `addToolApprovalResponse`: + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; +import { isToolUIPart, getToolName } from "ai"; + +function Chat() { + const agent = useAgent({ agent: "MyAgent" }); + const { messages, sendMessage, addToolApprovalResponse } = useAgentChat({ + agent + }); + + return ( +
+ {messages.map((message) => ( +
+ {message.parts?.map((part, i) => { + if (part.type === "text") { + return

{part.text}

; + } + + if (isToolUIPart(part)) { + // Tool waiting for approval + if ("approval" in part && part.state === "approval-requested") { + const approvalId = part.approval?.id; + return ( +
+

+ Approve {getToolName(part)} with{" "} + {JSON.stringify(part.input)}? +

+ + +
+ ); + } + + // Tool was denied + if (part.state === "output-denied") { + return ( +
{getToolName(part)}: Denied
+ ); + } + + // Tool completed + if (part.state === "output-available") { + return ( +
+ {getToolName(part)}: {JSON.stringify(part.output)} +
+ ); + } + } + + return null; + })} +
+ ))} +
+ ); +} +``` + +### Custom denial messages with `addToolOutput` + +When a user rejects a tool, `addToolApprovalResponse({ id, approved: false })` sets the tool state to `output-denied` with a generic "Tool execution denied." message. If you need to give the LLM a more specific reason for the denial, use `addToolOutput` with `state: "output-error"` instead: + +```tsx +const { addToolOutput } = useAgentChat({ agent }); + +// Reject with a custom error message +addToolOutput({ + toolCallId: part.toolCallId, + state: "output-error", + errorText: "User declined: insufficient budget for this quarter" +}); +``` + +This sends a `tool_result` to the LLM with your custom error text, so it can respond appropriately (e.g. suggest an alternative, ask clarifying questions). The `addToolOutput` function also works for tools in `approval-requested` or `approval-responded` states, not just `input-available`. + +`addToolApprovalResponse` (with `approved: false`) auto-continues the conversation when `autoContinueAfterToolResult` is enabled (the default), so the LLM sees the denial and can respond naturally. + +`addToolOutput` with `state: "output-error"` does **not** auto-continue — it gives you full control over what happens next. If you want the LLM to respond to the error, call `sendMessage()` afterward. + +See the complete example: [guides/human-in-the-loop/](../guides/human-in-the-loop/) + +## Client-Side Tool Execution with `onToolCall` + +For tools that need browser APIs (geolocation, camera, clipboard) or user interaction, define the tool on the server without an `execute` function and handle it on the client with `onToolCall`: + +### Server + +```typescript +export class MyAgent extends AIChatAgent { + async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages), + tools: { + // No execute function - client handles via onToolCall + getUserLocation: tool({ + description: "Get the user's current location from their browser", + inputSchema: z.object({}) + }) + }, + maxSteps: 3 + }); + + return result.toUIMessageStreamResponse(); + } +} +``` + +### Client + +```tsx +const { messages, sendMessage } = useAgentChat({ + agent, + onToolCall: async ({ toolCall, addToolOutput }) => { + if (toolCall.toolName === "getUserLocation") { + const position = await new Promise((resolve, reject) => { + navigator.geolocation.getCurrentPosition(resolve, reject); + }); + addToolOutput({ + toolCallId: toolCall.toolCallId, + output: { + lat: position.coords.latitude, + lng: position.coords.longitude + } + }); + } + } +}); +``` + +The server receives the tool output via `CF_AGENT_TOOL_RESULT` and can auto-continue the conversation (with `maxSteps > 1`), letting the LLM respond to the location data in the same turn. + +### OpenAI Agents SDK Pattern + +When using the [OpenAI Agents SDK](https://openai.github.io/openai-agents-js/), use the `needsApproval` function for conditional approval: + +```typescript +import { Agent } from "agents"; +import { tool, run } from "@openai/agents"; + +export class WeatherAgent extends Agent { + async processQuery(query: string) { + const weatherTool = tool({ + name: "get_weather", + description: "Get weather for a location", + parameters: z.object({ location: z.string() }), + + // Conditional approval - only for certain locations + needsApproval: async (_context, { location }) => { + return location === "San Francisco"; // Require approval for SF + }, + + execute: async ({ location }) => { + const conditions = ["sunny", "cloudy", "rainy"]; + return conditions[Math.floor(Math.random() * conditions.length)]; + } + }); + + const result = await run(this.openai, { + model: "gpt-4o", + tools: [weatherTool], + input: query + }); + + return result; + } +} +``` + +See the complete example: [openai-sdk/human-in-the-loop/](../openai-sdk/human-in-the-loop/) + +### MCP Elicitation + +When building MCP servers with `McpAgent`, you can request additional user input during tool execution using **elicitation**. The MCP client (like Claude Desktop) renders a form based on your JSON Schema and returns the user's response. + +```typescript +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { Agent } from "agents"; + +export class MyMcpAgent extends Agent { + server = new McpServer({ + name: "my-mcp-server", + version: "1.0.0" + }); + + onStart() { + this.server.registerTool( + "increase-counter", + { + description: "Increase the counter by a user-specified amount", + inputSchema: { + confirm: z.boolean().describe("Do you want to increase the counter?") + } + }, + async ({ confirm }, extra) => { + if (!confirm) { + return { content: [{ type: "text", text: "Cancelled." }] }; + } + + // Request additional input from the user + const userInput = await this.server.server.elicitInput( + { + message: "By how much do you want to increase the counter?", + requestedSchema: { + type: "object", + properties: { + amount: { + type: "number", + title: "Amount", + description: "The amount to increase the counter by" + } + }, + required: ["amount"] + } + }, + { relatedRequestId: extra.requestId } + ); + + // Check if user accepted or cancelled + if (userInput.action !== "accept" || !userInput.content) { + return { content: [{ type: "text", text: "Cancelled." }] }; + } + + // Use the input + const amount = Number(userInput.content.amount); + this.setState({ + ...this.state, + counter: this.state.counter + amount + }); + + return { + content: [ + { + type: "text", + text: `Counter increased by ${amount}, now at ${this.state.counter}` + } + ] + }; + } + ); + } +} +``` + +**Key differences from other patterns:** + +- Used by **MCP servers** exposing tools to clients, not agents calling tools +- Uses **JSON Schema** for structured form-based input +- The **MCP client** (Claude Desktop, etc.) handles UI rendering +- Returns `{ action: "accept" | "decline", content: {...} }` + +See the complete example: [examples/mcp-elicitation/](../examples/mcp-elicitation/) + +## State Patterns for Approvals + +Track pending approvals in agent state for UI rendering and persistence: + +```typescript +type PendingApproval = { + id: string; + workflowId?: string; + type: "expense" | "publish" | "delete"; + description: string; + amount?: number; + requestedBy: string; + requestedAt: number; + expiresAt?: number; +}; + +type ApprovalRecord = { + id: string; + approvalId: string; + decision: "approved" | "rejected"; + decidedBy: string; + decidedAt: number; + reason?: string; +}; + +type ApprovalState = { + pending: PendingApproval[]; + history: ApprovalRecord[]; +}; +``` + +### Multi-Approver Patterns + +For sensitive operations requiring multiple approvers: + +```typescript +type MultiApproval = { + id: string; + requiredApprovals: number; // e.g., 2 + currentApprovals: Array<{ + userId: string; + approvedAt: number; + }>; + rejections: Array<{ + userId: string; + rejectedAt: number; + reason: string; + }>; +}; + +@callable() +async approveMulti(approvalId: string, userId: string): Promise { + const approval = this.state.pending.find(p => p.id === approvalId); + if (!approval) throw new Error("Approval not found"); + + // Add this user's approval + approval.currentApprovals.push({ userId, approvedAt: Date.now() }); + + // Check if we have enough approvals + if (approval.currentApprovals.length >= approval.requiredApprovals) { + // Execute the approved action + await this.executeApprovedAction(approval); + return true; + } + + this.setState({ ...this.state }); + return false; // Still waiting for more approvals +} +``` + +## Timeouts and Escalation + +### Setting Approval Timeouts + +```typescript +const approval = await this.waitForApproval(step, { + timeout: "24 hours" +}); +``` + +### Escalation with Scheduling + +Use `schedule()` to set up escalation reminders: + +```typescript +@callable() +async submitForApproval(request: ApprovalRequest): Promise { + const approvalId = crypto.randomUUID(); + + // Add to pending + this.setState({ + ...this.state, + pending: [...this.state.pending, { id: approvalId, ...request }] + }); + + // Schedule reminder after 4 hours + await this.schedule( + Date.now() + 4 * 60 * 60 * 1000, + "sendReminder", + { approvalId } + ); + + // Schedule escalation after 24 hours + await this.schedule( + Date.now() + 24 * 60 * 60 * 1000, + "escalateApproval", + { approvalId } + ); + + return approvalId; +} +``` + +## Complete Examples + +| Pattern | Location | Description | +| ----------------- | ----------------------------------------------------------------- | -------------------------------------------------- | +| Workflow approval | [examples/workflows/](../examples/workflows/) | Multi-step task processing with approval gate | +| AIChatAgent tools | [guides/human-in-the-loop/](../guides/human-in-the-loop/) | Chat tool approval with needsApproval + onToolCall | +| OpenAI Agents SDK | [openai-sdk/human-in-the-loop/](../openai-sdk/human-in-the-loop/) | Conditional tool approval with modal | +| MCP Elicitation | [examples/mcp-elicitation/](../examples/mcp-elicitation/) | MCP server requesting structured user input | + +For detailed API documentation, see: + +- [Workflows](./workflows.md) - `waitForApproval()`, `approveWorkflow()`, `rejectWorkflow()` +- [MCP Servers](./mcp-servers.md) - `elicitInput()` for MCP elicitation +- [Callable Methods](./callable-methods.md) - `@callable()` decorator for approval endpoints diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 0000000000..ee1b58b8f4 --- /dev/null +++ b/docs/index.md @@ -0,0 +1,112 @@ +# Agents Documentation + +## Getting Started + +- [Getting Started](./getting-started.md) - Quick start guide for new users +- [Adding to an Existing Project](./adding-to-existing-project.md) - Integrate agents into your app +- [Understanding the Agent Class](./agent-class.md) - Deep dive into the Agent class architecture + +## Core Concepts + +- [State Management](./state.md) - Managing agent state with `setState()`, `initialState`, and `onStateChanged()` +- [Routing](./routing.md) - How `routeAgentRequest()` and agent naming works +- [HTTP & WebSockets](./http-websockets.md) - Request handling and real-time connections +- [Callable Methods](./callable-methods.md) - The `@callable` decorator and client-server method calls +- [Readonly Connections](./readonly-connections.md) - Restricting which connections can modify state +- [getCurrentAgent()](./get-current-agent.md) - Accessing agent context across async calls + +## Client SDK + +- [Client SDK](./client-sdk.md) - Connecting from React (`useAgent`) and vanilla JS (`AgentClient`), state sync, and RPC calls + +## Communication Channels + +- [Email Routing](./email.md) - Receiving and responding to emails +- [Webhooks](./webhooks.md) - Receiving and sending webhook events +- [Push Notifications](./push-notifications.md) - Browser push notifications via Web Push API and scheduled delivery +- TODO: [SMS](./sms.md) - Text message integration (Twilio, etc.) +- [Voice Agents](./voice.md) - Build voice agents with real-time speech-to-text, text-to-speech, and conversation persistence +- TODO: [Messengers](./messengers.md) - Slack, Discord, Telegram, and other chat platforms + +## Background Processing + +- [Queue](./queue.md) - Immediate background task execution +- [Scheduling](./scheduling.md) - Delayed, scheduled, and cron-based tasks +- [Retries](./retries.md) - Automatic retries with exponential backoff and jitter +- [Durable Execution](./durable-execution.md) - `runFiber()`, `stash()`, and crash recovery for long tasks +- [Workflows](./workflows.md) - Durable multi-step processing with Cloudflare Workflows +- [Human in the Loop](./human-in-the-loop.md) - Approval flows and manual intervention + +## AI Integration + +- TODO: [AI SDK Integration](./ai-sdk.md) - Using Vercel AI SDK with agents +- TODO: [TanStack Integration](./tanstack.md) - Using TanStack AI with agents +- [Chat Agents](./chat-agents.md) - `AIChatAgent` class and `useAgentChat` React hook +- [Server-Driven Messages](./server-driven-messages.md) - Autonomous agent workflows: scheduled follow-ups, queue processing, webhooks, chained reasoning +- TODO: [Using AI Models](./using-ai-models.md) - OpenAI, Anthropic, Workers AI, and other providers +- TODO: [RAG (Retrieval Augmented Generation)](./rag.md) - Vector search with Vectorize +- [Sessions (Experimental)](./sessions.md) - Persistent conversation storage with tree-structured messages, context blocks, compaction, and search +- [Workspace (Experimental)](./workspace.md) - Durable virtual filesystem backed by SQLite + R2 +- [Codemode (Experimental)](./codemode.md) - LLM-generated executable code for tool orchestration +- [Client Tools Continuation](./client-tools-continuation.md) - Handling tool calls across client/server +- [Resumable Streaming](./resumable-streaming.md) - Automatic stream resumption on disconnect + +## Think (Experimental) + +- [Overview](./think/index.md) - Opinionated chat agent with built-in memory, tools, and streaming +- [Getting Started](./think/getting-started.md) - Build your first Think agent step by step +- [Lifecycle Hooks](./think/lifecycle-hooks.md) - `beforeTurn`, `onStepFinish`, `onChunk`, `onChatResponse`, and more +- [Tools](./think/tools.md) - Workspace tools, code execution, extensions +- [Client Tools](./think/client-tools.md) - Browser-side tools, approvals, and concurrency +- [Sub-agents and Programmatic Turns](./think/sub-agents.md) - RPC streaming, `saveMessages`, recovery + +## MCP (Model Context Protocol) + +- [Creating MCP Servers](./mcp-servers.md) - Build MCP servers with `McpAgent` +- [Securing MCP Servers](./securing-mcp-servers.md) - OAuth and authentication for MCP +- [Connecting to MCP Servers](./mcp-client.md) - `addMcpServer()` and consuming external MCP tools +- [MCP Transports](./mcp-transports.md) - Transport options: Streamable HTTP, SSE, and RPC + +## Authentication & Security + +- TODO: [Securing your Agents](./securing-agents.md) - Authentication, authorization, and access control +- [Cross-Domain Authentication](./cross-domain-authentication.md) - Auth across different domains + +## Observability & Debugging + +- [Observability](./observability.md) - Monitoring and tracing agent activity +- TODO: [Testing](./testing.md) - Unit tests, integration tests, mocking agents +- TODO: [Evals](./evals.md) - Evaluating AI agent quality and behavior + +## Agent Studio + +- TODO: [Agent Studio](./agent-studio.md) - Local dev tool for inspecting and interacting with agent instances + +## Compute Environments + +- [Browse the Web (Experimental)](./browse-the-web.md) - Full CDP access for web inspection, scraping, and debugging +- TODO: [Cloudflare Sandboxes](./sandboxes.md) - Isolated environments for coding agents, ffmpeg, and heavy compute + +## Advanced Topics + +- [Long-Running Agents](./long-running-agents.md) - Building agents that persist for weeks or months: lifecycle, recovery, async operations, and planning +- TODO: [SQL API](./sql.md) - Using `this.sql` for direct database queries +- TODO: [Memory & Persistence](./memory.md) - Long-term storage patterns +- [Configuration](./configuration.md) - wrangler.jsonc setup, types, secrets, and deployment + +## Migration Guides + +- [Migration to AI SDK v5](./migration-to-ai-sdk-v5.md) +- [Migration to AI SDK v6](./migration-to-ai-sdk-v6.md) + +## Reference + +- TODO: [API Reference](./api-reference.md) - Complete API documentation +- TODO: [FAQ / How is this different from Durable Objects?](./faq.md) +- TODO: [Resources & Further Reading](./resources.md) + +--- + +## Contributing + +Found something missing? Documentation contributions are welcome! diff --git a/docs/long-running-agents.md b/docs/long-running-agents.md new file mode 100644 index 0000000000..6e452db920 --- /dev/null +++ b/docs/long-running-agents.md @@ -0,0 +1,713 @@ +# Long-Running Agents + +Build agents that persist for days, weeks, or months — surviving restarts, waking on demand, and managing work that spans far longer than any single request. + +## Why Cloudflare for long-running agents + +Agents spend most of their time waiting. Waiting for user input (seconds to days), LLM responses (seconds to minutes), tool results (seconds to hours), human approvals (hours to days), or scheduled wake-ups (minutes to months). On a traditional VM or container, you pay for all that idle time. An agent that is 99% dormant and 1% active still costs you 100% of a server. + +Durable Objects invert this model. An agent exists as an addressable entity with persistent state, but consumes zero compute when hibernated. When something happens — an HTTP request, a WebSocket message, a scheduled alarm, an inbound email — the platform wakes the agent, loads its state from SQLite, and hands it the event. The agent does its work, then goes back to sleep. + +This is the [actor model](https://en.wikipedia.org/wiki/Actor_model): each agent has an identity, durable state, and wakes on message. You do not manage servers, routing, health checks, or restart logic. The platform handles placement, scaling, and recovery. + +The economics follow directly: + +| | VMs / Containers | Durable Objects | +| --------------------------------------------- | ---------------------------------------------- | --------------------------------- | +| **Idle cost** | Full compute cost, always | Zero (hibernated) | +| **Scaling** | Provision and manage capacity | Automatic, per-agent | +| **State** | External database required | Built-in SQLite | +| **Recovery** | You build it (process managers, health checks) | Platform restarts, state survives | +| **Identity / routing** | You build it (load balancers, sticky sessions) | Built-in (name → agent) | +| **10,000 agents, each active 1% of the time** | 10,000 always-on instances | ~100 active at any moment | + +For agents — which are inherently bursty, stateful, and long-lived — this is a natural fit. + +## The lifecycle of a long-running agent + +A long-running agent is not a process that runs continuously. It is an entity that **exists** continuously but **runs** intermittently. Understanding the lifecycle is key to building agents that work reliably over long timelines. + +``` +Wake → onStart() → handle events → idle (~2 min) → hibernation + ▲ │ + └──────────────── alarm or request wakes agent ────────┘ + +Eviction (crash / redeploy) can happen at any point. +State persists in SQLite. Agent restarts on next event. +``` + +### What survives + +- **`this.state`** — persisted to SQLite on every `setState()` call +- **`this.sql` data** — all SQLite tables you create +- **Scheduled tasks** — stored in SQLite, trigger alarms to wake the agent +- **Connection state** — `connection.setState()` data for each WebSocket client +- **Fiber checkpoints** — `stash()` data from `runFiber()` + +Higher-level abstractions built on SQLite — such as [Workspace](./workspace.md) files, [Session](./sessions.md) messages, and MCP connection state — also survive, since they are backed by the same SQLite storage. + +### What does not survive + +- **In-memory variables** — class fields not stored via `setState()` or `this.sql` +- **Running timers** — `setTimeout`, `setInterval` are lost on hibernation/eviction +- **Open fetch requests** — in-flight HTTP calls are abandoned +- **Local closures** — callbacks and promise chains are lost + +The implication: any work that matters must be persisted or recoverable. The SDK provides primitives for this — schedules, fibers, queues — but understanding the boundary between "in-memory" and "durable" is essential. + +## Running example: a project manager agent + +Throughout this doc, we build up a project manager agent that: + +- Lives for the duration of a project (weeks or months) +- Tracks tasks, assigns work to sub-agents, and reports progress +- Wakes up on schedule to check deadlines and send reminders +- Reacts to external events (webhooks from GitHub, emails from team members) +- Handles long-running operations (CI pipelines, code reviews, deployments) +- Survives any number of restarts and evictions along the way + +```typescript +import { Agent } from "agents"; + +type ProjectState = { + name: string; + status: "planning" | "active" | "review" | "complete"; + tasks: Task[]; + plan: Plan | null; +}; + +type Task = { + id: string; + title: string; + status: "pending" | "in_progress" | "blocked" | "complete"; + assignee?: string; + dueDate?: string; + completedAt?: number; + externalJobId?: string; +}; + +export class ProjectManager extends Agent { + initialState: ProjectState = { + name: "", + status: "planning", + tasks: [], + plan: null + }; +} +``` + +The `Plan` type is introduced in [Planning as a durability strategy](#planning-as-a-durability-strategy). We add capabilities to this agent section by section. + +## Waking up: how agents get activated + +A hibernated agent can be woken by any of these sources: + +| Wake source | How it works | Example | +| ------------------------ | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------- | +| **HTTP request** | Any request to the agent's URL triggers `onRequest()` | A webhook from GitHub | +| **WebSocket connection** | A client connects, triggering `onConnect()` | A team member opens the dashboard | +| **RPC call** | Another Worker or agent calls a method via [service binding](https://developers.cloudflare.com/workers/runtime-apis/bindings/service-bindings/) or [`@callable`](./callable-methods.md) | A coordinator agent delegates a task | +| **Scheduled alarm** | A stored schedule fires, triggering `alarm()` → your callback | Daily standup reminder at 9am | +| **Email** | An inbound email triggers `email()` | A team member replies to a status email | + +The pattern extends naturally to any event source that can reach a Worker — anything from telephony webhooks to chat platform bots. An external signal arrives, the platform wakes the agent, and the agent handles it. + +The agent does not need to be "started" or "deployed" separately for each wake source — they all route to the same Durable Object instance. The agent's identity (its name) is the routing key. + +```typescript +export class ProjectManager extends Agent { + async onStart() { + // Daily deadline check at 9am UTC — idempotent, safe across restarts + await this.schedule( + "0 9 * * *", + "checkDeadlines", + {}, + { + idempotent: true + } + ); + + // Progress sync every 30 minutes + await this.scheduleEvery(1800, "syncProgress"); + } + + async onRequest(request: Request): Promise { + const url = new URL(request.url); + + if (url.pathname.endsWith("/github-webhook")) { + const event = await request.json(); + await this.handleGitHubEvent(event); + return new Response("OK"); + } + + return Response.json({ + project: this.state.name, + status: this.state.status + }); + } + + // Scheduled callbacks — the agent wakes, runs the method, goes back to sleep + async checkDeadlines() { + /* ... find overdue tasks, broadcast alerts ... */ + } + async syncProgress() { + /* ... check on sub-agents, update task statuses ... */ + } +} +``` + +## Staying alive during long work + +Sometimes an agent needs to do work that takes longer than the idle eviction window (~70–140 seconds). Streaming an LLM response, orchestrating a multi-step tool chain, or waiting on a slow API all risk the agent being evicted mid-flight. + +`keepAlive()` prevents this by creating a heartbeat that resets the inactivity timer: + +```typescript +export class ProjectManager extends Agent { + async generateProjectPlan(goal: string) { + const result = await this.keepAliveWhile(async () => { + const plan = await this.callLLM(`Create a project plan for: ${goal}`); + const tasks = await this.callLLM( + `Break this into tasks: ${JSON.stringify(plan)}` + ); + return { plan, tasks }; + }); + + this.setState({ + ...this.state, + status: "active", + plan: result.plan, + tasks: result.tasks + }); + } +} +``` + +`keepAliveWhile()` is the recommended approach — it guarantees the heartbeat is cleaned up when the work finishes (or throws). For manual control, `keepAlive()` returns a disposer: + +```typescript +const dispose = await this.keepAlive(); +try { + await longWork(); +} finally { + dispose(); +} +``` + +### When keepAlive is not enough + +`keepAlive` is for work measured in minutes, not hours. For truly long-running operations, use a different strategy: + +| Duration | Strategy | +| ---------------- | --------------------------------------------------------- | +| Seconds | Normal request handling | +| Minutes | `keepAlive()` / `keepAliveWhile()` | +| Minutes to hours | [Workflows](./workflows.md) | +| Hours to days | Async pattern: start job → hibernate → wake on completion | + +## Surviving crashes: fibers and recovery + +An agent can be evicted at any time — a deploy, a platform restart, or hitting resource limits. If the agent was mid-task, that work is lost unless it was checkpointed. + +[`runFiber()`](./durable-execution.md) provides crash-recoverable execution. It persists a row in SQLite for the duration of the work, and lets you `stash()` intermediate state. If the agent is evicted, the fiber row survives, and `onFiberRecovered()` is called on the next activation. + +```typescript +export class ProjectManager extends Agent { + async executeTask(task: Task) { + await this.runFiber(`task:${task.id}`, async (ctx) => { + const resources = await this.gatherResources(task); + ctx.stash({ phase: "prepared", resources, task }); + + const result = await this.runSubAgent(task, resources); + ctx.stash({ phase: "executed", result, task }); + + await this.updateTaskStatus(task.id, "complete", result); + }); + } + + async onFiberRecovered(ctx: FiberRecoveryContext) { + if (!ctx.name.startsWith("task:")) return; + const { phase, task } = ctx.snapshot as { phase: string; task: Task }; + + if (phase === "prepared") { + await this.executeTask(task); + } else if (phase === "executed") { + await this.updateTaskStatus( + task.id, + "complete", + (ctx.snapshot as { result: unknown }).result + ); + } + } +} +``` + +The pattern is: **checkpoint before expensive work, recover from the last checkpoint.** This is not automatic replay — you decide what recovery means for your domain. + +> **Testing recovery locally:** In `wrangler dev`, fiber recovery works identically to production. Kill the wrangler process (Ctrl-C or SIGKILL), restart it, and recovery fires automatically. If a request or WebSocket connection arrives first, `onStart()` runs `_checkRunFibers()` eagerly. If the agent has no incoming connections, the persisted alarm fires on its own and triggers recovery via `_onAlarmHousekeeping()` — this is critical for background agents that have no clients. Either path calls your `onFiberRecovered` hook. SQLite and alarm state persist to disk between restarts. + +For the full API reference — `FiberContext`, `FiberRecoveryContext`, concurrent fibers, inline vs fire-and-forget patterns — see [Durable Execution](./durable-execution.md). + +## Handling long async operations + +The project manager frequently kicks off work that takes far longer than any single activation — a CI pipeline runs for 20 minutes, a design review takes a day, a video asset takes hours to generate. The agent should not stay alive for any of this. Instead, it starts the work, persists the job ID in state, and hibernates. When the result arrives — via a callback, a poll, or a workflow completion — the agent wakes, correlates the result, and moves on. + +### Pattern: webhook callback + +The project manager starts a CI pipeline for a task. The pipeline takes 20 minutes. Rather than holding a connection open, the agent registers its own URL as the callback and goes to sleep: + +```typescript +export class ProjectManager extends Agent { + async startCIPipeline(task: Task) { + const response = await fetch("https://ci.example.com/api/pipelines", { + method: "POST", + body: JSON.stringify({ + repo: "org/project", + branch: "main", + callback_url: `${this.url}/ci-callback?taskId=${task.id}` + }) + }); + + const { pipelineId } = await response.json(); + this.updateTask(task.id, { + status: "in_progress", + externalJobId: pipelineId + }); + // Agent can now hibernate — it will wake when the CI service POSTs to the callback + } + + async onRequest(request: Request): Promise { + const url = new URL(request.url); + if (url.pathname.endsWith("/ci-callback")) { + const taskId = url.searchParams.get("taskId"); + const result = await request.json(); + this.updateTask(taskId, { + status: result.status === "success" ? "complete" : "blocked" + }); + return new Response("OK"); + } + // ... other routes + } +} +``` + +### Pattern: polling with schedule + +Not every external service supports callbacks. When the project manager submits a video asset for generation, it needs to check back periodically until the job completes: + +```typescript +export class ProjectManager extends Agent { + async startVideoGeneration(task: Task) { + const response = await fetch("https://video-api.example.com/generate", { + method: "POST", + body: JSON.stringify({ prompt: task.title }) + }); + const { jobId } = await response.json(); + this.updateTask(task.id, { status: "in_progress", externalJobId: jobId }); + await this.schedule(60, "pollExternalJob", { + taskId: task.id, + jobId, + attempt: 1 + }); + } + + async pollExternalJob(payload: { + taskId: string; + jobId: string; + attempt: number; + }) { + const response = await fetch( + `https://video-api.example.com/status/${payload.jobId}` + ); + const status = await response.json(); + + if (status.state === "complete" || status.state === "failed") { + this.updateTask(payload.taskId, { + status: status.state === "complete" ? "complete" : "blocked" + }); + return; + } + + // Still running — check again with backoff (max 10 minutes) + const nextDelay = Math.min(60 * payload.attempt, 600); + await this.schedule(nextDelay, "pollExternalJob", { + ...payload, + attempt: payload.attempt + 1 + }); + } +} +``` + +### Pattern: workflow delegation + +A production deployment involves multiple steps that must each retry independently — build, test, stage, promote. The project manager should not manage these steps internally; it delegates to a [Workflow](./workflows.md) that handles retries and step sequencing: + +```typescript +export class ProjectManager extends Agent { + async startDeployment(task: Task) { + const instanceId = await this.runWorkflow("DEPLOY_WORKFLOW", { + taskId: task.id, + environment: "production" + }); + this.updateTask(task.id, { + status: "in_progress", + externalJobId: instanceId + }); + } + + async onWorkflowComplete( + workflowName: string, + instanceId: string, + result?: unknown + ) { + const task = this.state.tasks.find((t) => t.externalJobId === instanceId); + if (task) this.updateTask(task.id, { status: "complete" }); + } +} +``` + +## Reconstructing context after a long wait + +The CI pipeline finishes 20 minutes later. The webhook wakes the project manager. The task status is updated. But now what? If the agent was using an LLM to orchestrate work — deciding which task to run next, drafting a status report, reasoning about blockers — it needs to pick up that reasoning thread. The original prompt, the in-flight tool call, the chain of thought — all gone from memory. + +This is the fundamental challenge of long-running AI agents. Most frameworks assume tool calls complete within the LLM's timeout and do not address this directly. + +Three approaches work today: + +**Replay the full conversation history.** `AIChatAgent` persists all messages in SQLite. When the result arrives, append it to the history and re-invoke the LLM. This is the simplest approach but re-processes the entire context window. + +**Stash a continuation summary.** Before hibernating, persist a compact description of what the agent was doing and what to do with the result: + +```typescript +ctx.stash({ + task: "Waiting for CI results", + onSuccess: "Mark task complete, move to next step in plan", + onFailure: "Notify team, schedule retry in 1 hour", + relevantContext: { taskId, planStep: 3 } +}); +``` + +On recovery, use the stash to construct a focused prompt rather than replaying everything. + +**Use the plan as context.** If the agent has a structured plan, the plan itself provides sufficient context: "I am on step 3 of 7, the step was 'run CI pipeline', the result just arrived." This is the most robust approach for long-running agents — the plan is both a recovery mechanism and a context reconstruction strategy. See the next section. + +## Planning as a durability strategy + +A structured plan is not just useful for showing progress to users — it is a durability mechanism. An agent with a plan can recover from any interruption by looking at where it left off. + +```typescript +type Plan = { + goal: string; + steps: PlanStep[]; + currentStep: number; + createdAt: string; + updatedAt: string; +}; + +type PlanStep = { + id: string; + description: string; + status: "pending" | "in_progress" | "complete" | "failed" | "skipped"; + result?: unknown; +}; + +export class ProjectManager extends Agent { + async createPlan(goal: string) { + const steps = await this.keepAliveWhile(async () => { + return this.callLLM(` + Break down this project goal into concrete steps. + Return a JSON array of { id, description } objects. + Goal: ${goal} + `); + }); + + this.setState({ + ...this.state, + plan: { + goal, + steps: steps.map((s: { id: string; description: string }) => ({ + ...s, + status: "pending" as const + })), + currentStep: 0, + createdAt: new Date().toISOString(), + updatedAt: new Date().toISOString() + } + }); + + await this.schedule(0, "executeNextStep"); + } + + async executeNextStep() { + const { plan } = this.state; + if (!plan || plan.currentStep >= plan.steps.length) { + this.setState({ ...this.state, status: "complete" }); + return; + } + + const step = plan.steps[plan.currentStep]; + + try { + const result = await this.keepAliveWhile(() => this.executeStep(step)); + + // Update plan state — advance to next step + const updatedSteps = plan.steps.map((s) => + s.id === step.id ? { ...s, status: "complete" as const, result } : s + ); + this.setState({ + ...this.state, + plan: { + ...plan, + steps: updatedSteps, + currentStep: plan.currentStep + 1, + updatedAt: new Date().toISOString() + } + }); + + // Schedule next step — the agent can hibernate between steps + await this.schedule(0, "executeNextStep"); + } catch (error) { + // Mark step failed — could re-plan, retry, or ask for human input + const updatedSteps = plan.steps.map((s) => + s.id === step.id ? { ...s, status: "failed" as const } : s + ); + this.setState({ + ...this.state, + plan: { + ...plan, + steps: updatedSteps, + updatedAt: new Date().toISOString() + } + }); + } + } +} +``` + +This pattern has several advantages for long-running agents: + +- **Recovery is trivial** — on restart, check `plan.currentStep` and resume +- **Progress is visible** — clients see which steps are done and what is next +- **Re-planning is possible** — if a step fails or requirements change, the agent can revise the remaining steps without losing completed work +- **Human oversight** — the plan is a natural approval checkpoint ("here is what I am going to do — proceed?") +- **Context reconstruction** — the plan tells the LLM where it is, what happened, and what to do next, without replaying the full conversation + +## Delegating to sub-agents + +A project manager does not do everything itself. It delegates specialized work to sub-agents — each with their own identity, state, and lifecycle. + +```typescript +export class ProjectManager extends Agent { + async delegateTask(task: Task) { + // Get a stub to a specialized agent (same DO namespace, unique name) + const researcher = await this.subAgent( + ResearchAgent, + `research-${task.id}` + ); + + // Call methods on the sub-agent via RPC — this wakes the sub-agent + const findings = await researcher.research(task.title); + + this.updateTask(task.id, { status: "complete" }); + return findings; + } +} +``` + +Sub-agents are independent Durable Objects. They have their own state, their own schedules, and their own lifecycle. The parent does not need to stay alive while the sub-agent works — it can start the work, hibernate, and be woken by a callback or scheduled check. + +For chat-oriented sub-agents, [Think](./think/index.md) provides `chat()` for RPC streaming between parent and child agents. See [Sub-agents and Programmatic Turns](./think/sub-agents.md). + +## Recovering interrupted LLM streams + +The patterns above handle the project manager's coordination work — scheduling, delegating, polling. But the project manager also uses an LLM directly: generating plans, summarizing progress, drafting status emails. Those LLM calls stream tokens over a connection that cannot be resumed if the agent is evicted mid-response. + +For chat-oriented agents built on `AIChatAgent`, this is an even sharper problem — the user is watching the response stream in real time and sees it stop mid-sentence. `chatRecovery` wraps each chat turn in a `runFiber`, providing automatic `keepAlive` during streaming and a recovery hook when the agent restarts: + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import type { + ChatRecoveryContext, + ChatRecoveryOptions +} from "@cloudflare/ai-chat"; + +class ProjectChat extends AIChatAgent { + override chatRecovery = true; + + override async onChatRecovery( + ctx: ChatRecoveryContext + ): Promise { + // ctx.partialText — text generated before eviction + // ctx.recoveryData — whatever you stashed via this.stash() + // ctx.messages — full conversation history + + // Default: persist partial response + schedule continuation + return {}; + } +} +``` + +The right recovery strategy depends on the LLM provider: + +| Provider | Strategy | How it works | Token cost | +| ---------------------- | ----------------------------------- | ----------------------------------------------------------------------------- | ---------- | +| Workers AI | Continue from partial | `continueLastTurn()` — model continues via assistant prefill | Low | +| OpenAI (Responses API) | Retrieve completed response | Stash `responseId` during streaming, `GET /v1/responses/{id}` on recovery | Zero | +| Anthropic | Synthetic continuation | Persist partial, send a synthetic user message asking the model to continue | Medium | +| Other | Try prefill, fall back to synthetic | `continueLastTurn()` if the provider supports it, synthetic message otherwise | Varies | + +For a complete multi-provider implementation with full code for each strategy, see the [`forever-chat` example](../experimental/forever-chat/) and the [`forever.md` design doc](../experimental/forever.md). + +[Think](./think/index.md) exposes `chatRecovery` as a configuration toggle — the recovery machinery is handled for you without implementing `onChatRecovery` yourself. + +## Managing state over time + +An agent that runs for months accumulates data: conversation history, timeline events, completed tasks, schedule records. Without management, this grows unbounded. + +### Housekeeping + +Schedule periodic cleanup to prune old data and archive completed work: + +```typescript +export class ProjectManager extends Agent { + async onStart() { + await this.schedule("0 0 * * *", "housekeeping", {}, { idempotent: true }); + } + + async housekeeping() { + // Archive completed tasks older than 30 days + const cutoff = Date.now() - 30 * 24 * 60 * 60 * 1000; + const toArchive = this.state.tasks.filter( + (t) => t.status === "complete" && (t.completedAt ?? 0) < cutoff + ); + for (const task of toArchive) { + this + .sql`INSERT INTO archived_tasks (id, data) VALUES (${task.id}, ${JSON.stringify(task)})`; + } + this.setState({ + ...this.state, + tasks: this.state.tasks.filter( + (t) => !toArchive.some((a) => a.id === t.id) + ) + }); + + // Clean up old workflow tracking records + this.deleteWorkflows({ + status: ["complete", "errored"], + createdBefore: new Date(Date.now() - 7 * 24 * 60 * 60 * 1000) + }); + } +} +``` + +### Conversation history management + +For agents that use `AIChatAgent`, conversation history can grow large over extended lifespans. Without management, a 3-month conversation will exhaust the LLM's context window long before the project ends. + +The [Session API](./sessions.md) addresses this directly: + +- **Compaction** — automatically summarizes older messages when the estimated token count exceeds a threshold. The summary replaces the middle of the conversation as a non-destructive overlay. Original messages remain in SQLite for audit. +- **Context blocks** — persistent structured sections injected into the system prompt (identity, memory, learned facts). The agent or the LLM can write to these blocks, and they survive hibernation and eviction. +- **Multi-session management** — `SessionManager` provides a registry of named sessions within a single agent, with forking, cross-session search, and `compactAndSplit` for splitting long conversations into linked continuations. + +For simpler cases: keep only the last N messages in the active context (sliding window), or selectively retain messages that contain decisions and approvals while pruning routine exchanges. + +## End of life + +A long-running agent eventually completes its purpose. The project ships, the investigation concludes, the monitoring window closes. Clean up explicitly: + +```typescript +export class ProjectManager extends Agent { + async completeProject() { + // Cancel remaining schedules + const schedules = this.getSchedules(); + for (const schedule of schedules) { + await this.cancelSchedule(schedule.id); + } + + // Archive final state + this.setState({ ...this.state, status: "complete" }); + + // Optionally destroy the Durable Object entirely + // All SQLite data, schedules, and state are permanently deleted + await this.destroy(); + } +} +``` + +`this.destroy()` is permanent. If you may need the agent's data later, archive it to an external store (R2, D1, or an API call) before destroying. For agents that might be reactivated, simply mark them as complete and let them hibernate — they cost nothing when idle. + +## When to use Workflows vs agent-internal patterns + +Both Workflows and agent-internal primitives (schedules, fibers, queues) support long-running work. The right choice depends on the nature of the work: + +| | Agent-internal | Workflows | +| ------------------ | ------------------------------------------------------ | ---------------------------------------- | +| **Best for** | Agent-centric work: scheduling, polling, state updates | Independent multi-step pipelines | +| **Durability** | SQLite (survives eviction) | Workflow engine (survives everything) | +| **Retries** | `this.retry()`, schedule-level retries | Per-step retries with backoff | +| **Max duration** | Minutes per activation (with `keepAlive`) | 30 minutes per step, unlimited steps | +| **Human approval** | Build it yourself (state + WebSocket) | Built-in `waitForApproval()` | +| **Complexity** | Lower — everything is in the agent | Higher — separate class, wrangler config | + +A pragmatic rule: if the work is about the agent managing its own lifecycle (checking deadlines, syncing state, sending reminders), use schedules and fibers. If the work is a discrete pipeline that could fail and retry independently (deploy, data processing, report generation), use a Workflow. + +The project manager agent uses both: schedules for its own rhythms (daily standups, progress syncs), and Workflows for heavyweight operations (deployments, CI pipelines). + +## Think: batteries included + +If you are building a chat-oriented long-running agent and want these patterns built in rather than assembling them yourself, [`Think`](./think/index.md) provides them out of the box: + +- **Sessions with compaction** — non-destructive conversation summarization, context blocks, cross-session search +- **Fiber-based recovery** — `chatRecovery` as a configuration toggle +- **Sub-agent RPC** — `chat()` for parent-child streaming +- **Persistent memory** — LLM-writable context blocks that survive hibernation +- **Workspace and code execution** — built-in file tools and sandboxed execution + +Override `getModel()` and `configureSession()` and the durability machinery is handled for you. Think is the opinionated path; the primitives described in this doc are what Think is built on. + +## Summary + +Long-running agents on Cloudflare are not long-running processes. They are durable entities that wake, work, and sleep — potentially over weeks or months. The key primitives: + +| Primitive | Purpose | +| -------------------------------------- | -------------------------------------------------------------- | +| **`setState()` / `this.sql`** | Persist state across activations | +| **`schedule()` / `scheduleEvery()`** | Wake the agent at future times | +| **`keepAlive()` / `keepAliveWhile()`** | Prevent eviction during active work | +| **`runFiber()` / `stash()`** | Checkpoint and recover long tasks | +| **`chatRecovery`** | Recover interrupted LLM streams | +| **`onRequest()` / `email()` / RPC** | Wake on external events | +| **`runWorkflow()`** | Delegate heavyweight multi-step work | +| **`subAgent()`** | Delegate specialized work to child agents | +| **Session API** | Manage conversation history, compaction, and context over time | +| **Structured plans in state** | Enable recovery, visibility, and re-planning | + +For the project manager agent, these compose into an agent that: + +1. **Plans** — breaks goals into steps, persists the plan in state +2. **Executes** — runs steps one at a time, hibernating between them +3. **Reacts** — wakes on webhooks, emails, and schedules +4. **Recovers** — resumes from the last checkpoint after any interruption +5. **Delegates** — hands off work to sub-agents and Workflows +6. **Maintains** — prunes old data, archives completed work, manages its own lifecycle +7. **Ends** — cleans up and destroys itself when the project is done + +The agent does not need to run continuously to do any of this. It just needs to exist. + +## Related + +- [Durable Execution](./durable-execution.md) — `runFiber()`, `stash()`, and crash recovery +- [Scheduling](./scheduling.md) — delayed, cron, and interval tasks +- [Retries](./retries.md) — retry options and patterns +- [Workflows](./workflows.md) — durable multi-step processing +- [State Management](./state.md) — `setState()` and persistence +- [Sessions](./sessions.md) — persistent conversation storage, compaction, and context blocks +- [Think](./think/index.md) — opinionated chat agent with built-in durability +- [HTTP & WebSockets](./http-websockets.md) — lifecycle hooks and hibernation +- [Callable Methods](./callable-methods.md) — RPC via `@callable` and service bindings +- [Email Routing](./email.md) — receiving inbound email +- [Webhooks](./webhooks.md) — receiving external events +- [Human in the Loop](./human-in-the-loop.md) — approval flows +- [Resumable Streaming](./resumable-streaming.md) — client-side stream resumption on disconnect +- [`forever-chat` example](../experimental/forever-chat/) — multi-provider LLM recovery demo diff --git a/docs/mcp-client.md b/docs/mcp-client.md new file mode 100644 index 0000000000..0cf706665e --- /dev/null +++ b/docs/mcp-client.md @@ -0,0 +1,620 @@ +# Connecting to MCP Servers + +Connect your agent to external MCP (Model Context Protocol) servers to use their tools, resources, and prompts. This enables your agent to interact with GitHub, Slack, databases, and other services through a standardized protocol. + +## Overview + +The MCP client capability lets your agent: + +- **Connect to external MCP servers** - GitHub, Slack, databases, AI services +- **Use their tools** - Call functions exposed by MCP servers +- **Access resources** - Read data from MCP servers +- **Use prompts** - Leverage pre-built prompt templates + +> **Note:** This page covers connecting to MCP servers as a client. To create your own MCP server, see [Creating MCP Servers](./mcp-servers.md). + +## Quick Start + +```typescript +import { Agent } from "agents"; + +export class MyAgent extends Agent { + async onRequest(request: Request) { + // Add an MCP server + const result = await this.addMcpServer( + "github", + "https://mcp.github.com/mcp" + ); + + if (result.state === "authenticating") { + // Server requires OAuth - redirect user to authorize + return Response.redirect(result.authUrl); + } + + // Server is ready - tools are now available + const state = this.getMcpServers(); + console.log(`Connected! ${state.tools.length} tools available`); + + return new Response("MCP server connected"); + } +} +``` + +## Adding MCP Servers + +Use `addMcpServer()` to connect to an MCP server: + +```typescript +const result = await this.addMcpServer(name, url, options?); +``` + +### Basic Usage + +```typescript +// Simple connection +await this.addMcpServer("notion", "https://mcp.notion.so/mcp"); + +// With explicit callback host (rarely needed — auto-derived from request or WebSocket URI) +await this.addMcpServer("github", "https://mcp.github.com/mcp", { + callbackHost: "https://my-worker.workers.dev" +}); +``` + +### Transport Options + +MCP supports multiple transport types: + +```typescript +await this.addMcpServer("server", "https://mcp.example.com/mcp", { + transport: { + // Transport type: "streamable-http" (default), "sse", or "auto" + type: "streamable-http" + } +}); +``` + +| Transport | Description | +| ------------------- | ----------------------------------------------------- | +| `"streamable-http"` | HTTP with streaming - recommended default | +| `"sse"` | Server-Sent Events - legacy / compatibility transport | +| `"auto"` | Auto-detect based on server response | + +### Custom Headers + +For servers behind authentication (like Cloudflare Access) or using bearer tokens: + +```typescript +await this.addMcpServer("internal", "https://internal-mcp.example.com/mcp", { + transport: { + headers: { + Authorization: "Bearer my-token", + "CF-Access-Client-Id": "...", + "CF-Access-Client-Secret": "..." + } + } +}); +``` + +### Retry Options + +Configure retry behavior for connection and reconnection attempts: + +```typescript +await this.addMcpServer("github", "https://mcp.github.com/mcp", { + retry: { + maxAttempts: 5, + baseDelayMs: 1000, + maxDelayMs: 10000 + } +}); +``` + +These options are persisted and used when reconnecting after hibernation or after OAuth completion. Default: 3 attempts, 500ms base delay, 5s max delay. See [Retries](./retries.md) for more details. + +### URL Security + +MCP server URLs are validated before connection to prevent Server-Side Request Forgery (SSRF). The following URL targets are blocked: + +- Private/internal IP ranges (RFC 1918: `10.x`, `172.16-31.x`, `192.168.x`) +- Unspecified addresses (`0.0.0.0`, `::`) +- Link-local addresses (`169.254.x`, `fe80::`) +- Cloud metadata endpoints (`169.254.169.254`) +- IPv6 unique-local addresses (`fc00::/7`) + +Loopback development URLs such as `localhost`, `127.0.0.1`, and `::1` are allowed. + +If you need to connect to another internal MCP server, use the [RPC transport](./mcp-transports.md) with a Durable Object binding instead of HTTP. + +### Return Value + +`addMcpServer()` returns the connection state: + +```typescript +type AddMcpServerResult = + | { id: string; state: "ready" } + | { id: string; state: "authenticating"; authUrl: string }; +``` + +- **`ready`** - Server connected and tools discovered +- **`authenticating`** - Server requires OAuth; redirect user to `authUrl` + +## OAuth Authentication + +Many MCP servers require OAuth authentication. The agent handles the OAuth flow automatically. + +### How It Works + +```mermaid +sequenceDiagram + participant Client + participant Agent + participant MCPServer + + Client->>Agent: addMcpServer(name, url) + Agent->>MCPServer: Connect + MCPServer-->>Agent: Requires OAuth + Agent-->>Client: state: authenticating, authUrl + Client->>MCPServer: User authorizes + MCPServer->>Agent: Callback with code + Agent->>MCPServer: Exchange for token + Agent-->>Client: onMcpUpdate (ready) +``` + +### Handling OAuth in Your Agent + +```typescript +async onRequest(request: Request) { + const result = await this.addMcpServer("github", "https://mcp.github.com/mcp"); + + if (result.state === "authenticating") { + // Option 1: Redirect the user + return Response.redirect(result.authUrl); + + // Option 2: Return the URL for client-side redirect + return Response.json({ + status: "needs_auth", + authUrl: result.authUrl + }); + } + + return Response.json({ status: "connected", id: result.id }); +} +``` + +### OAuth Callback + +The callback URL is automatically constructed: + +``` +https://{host}/{agentsPrefix}/{agent-name}/{instance-name}/callback +``` + +For example: `https://my-worker.workers.dev/agents/my-agent/default/callback` + +OAuth tokens are securely stored in SQLite and persist across agent restarts. + +### Custom Callback Handling + +For custom OAuth completion behavior: + +```typescript +// In your agent constructor or onStart +this.mcp.configureOAuthCallback({ + // Redirect after successful auth + successRedirect: "https://myapp.com/success", + + // Redirect on error + errorRedirect: "https://myapp.com/error", + + // Or use a custom handler + customHandler: (result) => { + return new Response( + JSON.stringify({ + success: result.authSuccess, + serverId: result.serverId, + error: result.authError + }), + { + headers: { "Content-Type": "application/json" } + } + ); + } +}); +``` + +### Custom OAuth Provider + +By default, agents use dynamic client registration to authenticate with MCP servers. If you need to use a different OAuth strategy — such as pre-registered client credentials, mTLS-based authentication, or other mechanisms — override the `createMcpOAuthProvider` method in your agent subclass: + +```typescript +import { Agent } from "agents"; +import type { AgentMcpOAuthProvider } from "agents"; + +class MyAgent extends Agent { + createMcpOAuthProvider(callbackUrl: string): AgentMcpOAuthProvider { + return new MyCustomOAuthProvider(this.ctx.storage, this.name, callbackUrl); + } +} +``` + +Your custom class must implement the `AgentMcpOAuthProvider` interface, which extends the MCP SDK's `OAuthClientProvider` with additional properties (`authUrl`, `clientId`, `serverId`) and methods (`checkState`, `consumeState`, `deleteCodeVerifier`) used by the agent's MCP connection lifecycle. + +The override is used for both new connections (`addMcpServer`) and restored connections after a Durable Object restart, so your custom provider is always used consistently. + +#### Custom storage backend + +The most common customization is using a different storage backend while keeping the built-in OAuth logic (CSRF state, PKCE, nonce generation, token management). Import `DurableObjectOAuthClientProvider` and pass your own storage adapter: + +```typescript +import { Agent, DurableObjectOAuthClientProvider } from "agents"; +import type { AgentMcpOAuthProvider } from "agents"; + +class MyAgent extends Agent { + createMcpOAuthProvider(callbackUrl: string): AgentMcpOAuthProvider { + return new DurableObjectOAuthClientProvider( + myCustomStorage, // any DurableObjectStorage-compatible adapter + this.name, + callbackUrl + ); + } +} +``` + +## Using MCP Capabilities + +Once connected, access the server's capabilities: + +### Getting Available Tools + +```typescript +const state = this.getMcpServers(); + +// All tools from all connected servers +for (const tool of state.tools) { + console.log(`Tool: ${tool.name}`); + console.log(` From server: ${tool.serverId}`); + console.log(` Description: ${tool.description}`); +} +``` + +### Resources and Prompts + +```typescript +const state = this.getMcpServers(); + +// Available resources +for (const resource of state.resources) { + console.log(`Resource: ${resource.name} (${resource.uri})`); +} + +// Available prompts +for (const prompt of state.prompts) { + console.log(`Prompt: ${prompt.name}`); +} +``` + +### Server Status + +```typescript +const state = this.getMcpServers(); + +for (const [id, server] of Object.entries(state.servers)) { + console.log(`${server.name}: ${server.state}`); + // state: "ready" | "authenticating" | "connecting" | "connected" | "discovering" | "failed" +} +``` + +### Integration with AI SDK + +To use MCP tools with the Vercel AI SDK, use `this.mcp.getAITools()` which converts MCP tools to AI SDK format: + +```typescript +import { generateText } from "ai"; + +async function chat(prompt: string) { + const response = await generateText({ + model: openai("gpt-4"), + prompt, + tools: this.mcp.getAITools() // Converts MCP tools to AI SDK format + }); + + return response; +} +``` + +> **Note:** `getMcpServers().tools` returns raw MCP `Tool` objects for inspection. Use `this.mcp.getAITools()` when passing tools to the AI SDK. + +## Managing Servers + +### Removing a Server + +```typescript +await this.removeMcpServer(serverId); +``` + +This disconnects from the server and removes it from storage. + +### Persistence + +MCP servers persist across agent restarts: + +- Server configuration stored in SQLite +- OAuth tokens stored securely +- Connections restored automatically when agent wakes + +### Listing All Servers + +```typescript +const state = this.getMcpServers(); + +for (const [id, server] of Object.entries(state.servers)) { + console.log(`${id}: ${server.name} (${server.server_url})`); +} +``` + +## Client-Side Integration + +Connected clients receive real-time MCP updates via WebSocket: + +```typescript +import { useAgent } from "agents/react"; + +function Dashboard() { + const [tools, setTools] = useState([]); + const [servers, setServers] = useState({}); + + const agent = useAgent({ + agent: "MyAgent", + onMcpUpdate: (mcpState) => { + setTools(mcpState.tools); + setServers(mcpState.servers); + } + }); + + return ( +
+

Connected Servers

+ {Object.entries(servers).map(([id, server]) => ( +
+ {server.name}: {server.connectionState} +
+ ))} + +

Available Tools ({tools.length})

+ {tools.map(tool => ( +
+ {tool.name} +
+ ))} +
+ ); +} +``` + +## Advanced: MCPClientManager + +For fine-grained control, use `this.mcp` directly: + +### Step-by-Step Connection + +```typescript +// 1. Register the server (saves to storage and creates in-memory connection) +const id = "my-server"; +await this.mcp.registerServer(id, { + url: "https://mcp.example.com/mcp", + name: "My Server", + callbackUrl: "https://my-worker.workers.dev/agents/my-agent/default/callback", + transport: { type: "auto" } +}); + +// 2. Connect (initializes transport, handles OAuth if needed) +const connectResult = await this.mcp.connectToServer(id); + +if (connectResult.state === "failed") { + console.error("Connection failed:", connectResult.error); + return; +} + +if (connectResult.state === "authenticating") { + console.log("OAuth required:", connectResult.authUrl); + return; +} + +// 3. Discover capabilities (transitions from "connected" to "ready") +if (connectResult.state === "connected") { + const discoverResult = await this.mcp.discoverIfConnected(id); + + if (!discoverResult?.success) { + console.error("Discovery failed:", discoverResult?.error); + } +} +``` + +### Event Subscription + +```typescript +// Listen for state changes (onServerStateChanged is an Event) +const disposable = this.mcp.onServerStateChanged(() => { + console.log("MCP server state changed"); + this.broadcastMcpServers(); // Notify connected clients +}); + +// Clean up the subscription when no longer needed +// disposable.dispose(); +``` + +### Waiting for Connections + +After hibernation or when connections are being restored in the background, MCP tools may not be immediately available. Use `waitForConnections()` to wait until all in-flight connection and discovery operations have settled: + +```typescript +// Wait indefinitely for all connections to be ready +await this.mcp.waitForConnections(); + +// Wait with a timeout (in milliseconds) +await this.mcp.waitForConnections({ timeout: 10_000 }); +``` + +This is useful when you need to call `this.mcp.getAITools()` immediately after the agent wakes from hibernation. Without waiting, tools from servers that are still reconnecting will be missing. + +> **Note:** `AIChatAgent` handles this automatically via the `waitForMcpConnections` property (defaults to `{ timeout: 10_000 }`). You only need `waitForConnections()` directly when using `Agent` with MCP, or when you want finer control inside `onChatMessage`. + +### Error Recovery + +```typescript +async retryConnection(serverId: string) { + const result = await this.mcp.connectToServer(serverId); + + if (result.state === "connected") { + await this.mcp.discoverIfConnected(serverId); + } else if (result.state === "failed") { + console.error("Reconnection failed:", result.error); + } +} +``` + +## Examples + +### MCP Client Demo + +The [`examples/mcp-client`](https://github.com/cloudflare/agents/tree/main/examples/mcp-client) example demonstrates: + +- Adding and removing MCP servers dynamically +- Custom OAuth callback handling (popup-closing behavior) +- Listing tools from connected servers +- Real-time state updates to the frontend + +```typescript +// From examples/mcp-client/src/server.ts +export class MyAgent extends Agent { + onStart() { + // Custom OAuth callback that closes the popup window + this.mcp.configureOAuthCallback({ + customHandler: (result) => { + if (result.authSuccess) { + return new Response("", { + headers: { "content-type": "text/html" } + }); + } + // Handle error... + } + }); + } + + async onRequest(request: Request) { + const url = new URL(request.url); + + if (url.pathname.endsWith("add-mcp")) { + const { name, url } = await request.json(); + await this.addMcpServer(name, url); + return new Response("Ok"); + } + // ... + } +} +``` + +## API Reference + +### addMcpServer() + +```typescript +// HTTP transport (Streamable HTTP, SSE) +async addMcpServer( + name: string, + url: string, + options?: { + callbackHost?: string; // auto-derived from request or WebSocket connection URI; only set to override + callbackPath?: string; // custom callback URL path (bypasses default /agents/{class}/{name}/callback) + agentsPrefix?: string; + client?: ClientOptions; + transport?: { + headers?: HeadersInit; + type?: "sse" | "streamable-http" | "auto"; // default: "auto" + }; + retry?: RetryOptions; // retry options for connection/reconnection + } +): Promise< + | { id: string; state: "ready" } + | { id: string; state: "authenticating"; authUrl: string } +> + +// RPC transport (Durable Object binding — no HTTP overhead) +async addMcpServer( + name: string, + binding: DurableObjectNamespace, + options?: { + props?: Record; // passed to the McpAgent's onStart(props) + client?: ClientOptions; + retry?: RetryOptions; + } +): Promise<{ id: string; state: "ready" }> + +// Legacy signature (still supported) +async addMcpServer( + name: string, + url: string, + callbackHost?: string, + agentsPrefix?: string, + options?: { ... } +): Promise<...> +``` + +Add and connect to an MCP server. Throws if connection or discovery fails. + +`callbackHost` is automatically derived from the incoming HTTP request or WebSocket connection URI — you almost never need to set it explicitly. It is only needed when the auto-detected host does not match your desired OAuth callback origin (for example, behind a reverse proxy). For RPC transport, pass a `DurableObjectNamespace` binding instead of a URL. See [MCP Transports](./mcp-transports.md) for details. + +Calling `addMcpServer` is idempotent when both the server name **and** URL match an existing active connection — the existing connection is returned without creating a duplicate. This makes it safe to call in `onStart()` without worrying about duplicate connections on restart. + +If you call `addMcpServer` with the same name but a **different** URL, a new connection is created. Both connections remain active and their tools are merged in `getAITools()`. To replace a server, call `removeMcpServer(oldId)` first. + +> **Note:** URLs are normalized before comparison (trailing slashes, default ports, and hostname case are handled), so `https://MCP.Example.com` and `https://mcp.example.com/` are treated as the same URL. + +### removeMcpServer() + +```typescript +async removeMcpServer(id: string): Promise +``` + +Disconnect from and remove an MCP server. + +### getMcpServers() + +```typescript +getMcpServers(): MCPServersState +``` + +Get the current state of all MCP servers and their capabilities. + +### MCPServersState + +```typescript +type MCPServersState = { + servers: { + [id: string]: MCPServer; + }; + tools: (Tool & { serverId: string })[]; + prompts: (Prompt & { serverId: string })[]; + resources: (Resource & { serverId: string })[]; +}; +``` + +### MCPServer + +```typescript +type MCPServer = { + name: string; + server_url: string; + auth_url: string | null; + state: + | "ready" + | "authenticating" + | "connecting" + | "connected" + | "discovering" + | "failed"; + error: string | null; + instructions: string | null; + capabilities: ServerCapabilities | null; +}; +``` diff --git a/docs/mcp-servers.md b/docs/mcp-servers.md new file mode 100644 index 0000000000..8ac784e43d --- /dev/null +++ b/docs/mcp-servers.md @@ -0,0 +1,526 @@ +# Creating MCP Servers + +This guide covers the different ways to create MCP servers with the Agents SDK and helps you choose the right approach. + +## Choosing an Approach + +| Approach | Stateful? | Requires Durable Objects? | Best for | +| ---------------------------------------------- | --------- | ------------------------- | ---------------------------------------------- | +| `createMcpHandler()` | No | No | Stateless tools, simplest setup | +| `McpAgent` | Yes | Yes | Stateful tools, per-session state, elicitation | +| Raw `WebStandardStreamableHTTPServerTransport` | No | No | Full control, no SDK dependency | + +- **`createMcpHandler()`** is the fastest way to get a stateless MCP server running. Use it when your tools do not need per-session state. +- **`McpAgent`** gives you a Durable Object per session with built-in state management, elicitation support, and both SSE and Streamable HTTP transports. +- **Raw transport** gives you full control if you want to use the `@modelcontextprotocol/sdk` directly without the Agents SDK helpers. + +## Stateless MCP Server with `createMcpHandler()` + +The simplest way to create an MCP server. No Durable Objects or bindings required: + +```typescript +import { createMcpHandler } from "agents/mcp"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { z } from "zod"; + +function createServer() { + const server = new McpServer({ + name: "Hello MCP Server", + version: "1.0.0" + }); + + server.registerTool( + "hello", + { + description: "Returns a greeting message", + inputSchema: { name: z.string().optional() } + }, + async ({ name }) => ({ + content: [{ text: `Hello, ${name ?? "World"}!`, type: "text" }] + }) + ); + + return server; +} + +export default { + fetch: async (request: Request, env: Env, ctx: ExecutionContext) => { + const server = createServer(); + return createMcpHandler(server)(request, env, ctx); + } +}; +``` + +> **Important:** Create a new `McpServer` instance per request. The MCP SDK does not allow connecting an already-connected server to a new transport. + +### `createMcpHandler` Options + +```typescript +createMcpHandler(server, { + route: "/mcp", // path to handle (default: "/mcp") + enableJsonResponse: true, // use JSON responses instead of SSE streaming + sessionIdGenerator: () => crypto.randomUUID(), + corsOptions: { ... }, // CORS configuration + authContext: { props: {} }, // manually set auth context + transport: workerTransport // provide your own WorkerTransport instance +}); +``` + +### Accessing Authenticated User Context + +When your MCP server is wrapped with `OAuthProvider` from `@cloudflare/workers-oauth-provider`, authenticated user information is available inside tools via `getMcpAuthContext()`: + +```typescript +import { createMcpHandler, getMcpAuthContext } from "agents/mcp"; + +server.registerTool( + "whoami", + { description: "Returns the authenticated user" }, + async () => { + const auth = getMcpAuthContext(); + return { + content: [ + { + type: "text", + text: auth ? JSON.stringify(auth.props) : "Not authenticated" + } + ] + }; + } +); +``` + +The `OAuthProvider` sets `ctx.props` on the execution context, which `createMcpHandler` automatically picks up and makes available via `getMcpAuthContext()`. + +## Stateful MCP Server with `McpAgent` + +`McpAgent` gives each client session its own Durable Object with persistent state. Use this when your tools need to track per-session data. + +### Writing TinyMCP + +Prototyping is very easy! If you want to quickly deploy an MCP, it only takes ~20 lines of code: + +```typescript +import { McpAgent } from "agents/mcp"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { z } from "zod"; + +// Our MCP server! +export class TinyMcp extends McpAgent { + server = new McpServer({ name: "", version: "v1.0.0" }); + + async init() { + this.server.registerTool( + "square", + { + description: "Squares a number", + inputSchema: { number: z.number() } + }, + async ({ number }) => ({ + content: [{ type: "text", text: String(number ** 2) }] + }) + ); + } +} + +// This is literally all there is to our Worker +export default TinyMcp.serve("/"); +``` + +Your `wrangler.jsonc` would look something like: + +```jsonc +{ + "name": "tinymcp", + "main": "src/index.ts", + "compatibility_date": "2026-01-28", + "compatibility_flags": ["nodejs_compat"], + "durable_objects": { + "bindings": [ + { + "name": "MCP_OBJECT", + "class_name": "TinyMcp" + } + ] + }, + "migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["TinyMcp"] + } + ] +} +``` + +### What is going on here? + +`McpAgent` requires us to define 2 bits, `server` and `init()`. + +`init()` is the initialization logic that runs every time our MCP server is started (each client session goes to a different Agent instance). +In there you'll normally setup all your tools/resources and anything else you might need. In this case, we're only setting the tool `square`. + +That was just the `McpAgent`, but we still need a Worker to route requests to our MCP server. `McpAgent` exports a static method that deals with that for you. That's what `TinyMcp.serve(...)` is for. +It returns an object with a `fetch` handler that can act as our Worker entrypoint and deal with the Streamable HTTP transport for us, so we can deploy our MCP directly! + +### Putting it to the test + +It's a very simple MCP indeed, but you can get a feel of how fast you can get a server up and running. You can deploy this worker and test your MCP with any client. I'll try with https://playground.ai.cloudflare.com: +![model calls the square tool after connecting to our mcp](https://github.com/user-attachments/assets/1e979a82-ed3e-49e9-b9d5-a3fc9b0363a7) + +## Password-protected StorageMcp with OAuth! + +To get a feel of what a more realistic MCP might look like, let's deploy an MCP that lets anyone that knows our secret password access a shared R2 bucket. (This is an example of a custom authorization flow, please do **not** use this in production) + +```typescript +import { McpAgent } from "agents/mcp"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { + OAuthProvider, + type OAuthHelpers +} from "@cloudflare/workers-oauth-provider"; +import { z } from "zod"; +import { env } from "cloudflare:workers"; + +export class StorageMcp extends McpAgent { + server = new McpServer({ name: "", version: "v1.0.0" }); + + async init() { + // Helper to return text responses from our tools + const textRes = (text: string) => ({ + content: [{ type: "text" as const, text }] + }); + + this.server.registerTool( + "writeFile", + { + description: "Store text as a file with the given path", + inputSchema: { + path: z.string().describe("Absolute path of the file"), + content: z.string().describe("The content to store") + } + }, + async ({ path, content }) => { + try { + await env.BUCKET.put(path, content); + return textRes(`Successfully stored contents to ${path}`); + } catch (e: unknown) { + return textRes(`Couldn't save to file. Found error ${e}`); + } + } + ); + + this.server.registerTool( + "readFile", + { + description: "Read the contents of a file", + inputSchema: { + path: z.string().describe("Absolute path of the file to read") + } + }, + async ({ path }) => { + const obj = await env.BUCKET.get(path); + if (!obj || !obj.body) + return textRes(`Error reading file at ${path}: not found`); + try { + return textRes(await obj.text()); + } catch (e: unknown) { + return textRes(`Error reading file at ${path}: ${e}`); + } + } + ); + + this.server.registerTool( + "whoami", + { + description: "Check who the user is" + }, + async () => { + return textRes(`${this.props?.userId}`); + } + ); + } +} + +// HTML form page for users to write our password +function passwordPage(opts: { query: string; error?: string }) { + const err = opts.error + ? `

${opts.error}

` + : ""; + return new Response( + ` + + + + + ENTER THE MAGIC WORD + + + +
+

ENTER THE MAGIC WORD

+ ${err} + + + +
+ +`, + { headers: { "content-type": "text/html; charset=utf-8" } } + ); +} + +// This is the default handler of our worker BEFORE requests are authenticated. +interface StorageEnv { + OAUTH_PROVIDER: OAuthHelpers; + SHARED_PASSWORD: string; +} + +const defaultHandler = { + async fetch(request: Request, env: StorageEnv) { + const provider = env.OAUTH_PROVIDER; + const url = new URL(request.url); + + // Only handle our auth UI/flow here + if (url.pathname !== "/authorize") { + return new Response("NOT FOUND", { status: 404 }); + } + + // Parse the OAuth request + const oauthReq = await provider.parseAuthRequest(request); + + // We render the password page for GET requests + if (request.method === "GET") { + return passwordPage({ query: url.searchParams.toString() }); + } + + // We validate the password in POST requests + if (request.method === "POST") { + const form = await request.formData(); + const password = String(form.get("password") || ""); + + const SHARED_PASSWORD = env.SHARED_PASSWORD; // Store this as a secret + if (!SHARED_PASSWORD) { + return new Response("Server misconfigured: missing SHARED_PASSWORD", { + status: 500 + }); + } + if (password !== SHARED_PASSWORD) { + return passwordPage({ + query: url.searchParams.toString(), + error: "Wrong password." + }); + } + + // We give everyone the same userId + const userId = "friend"; + + const { redirectTo } = await provider.completeAuthorization({ + request: oauthReq, + userId, + scope: [], // We don't care about scopes + + // We could add anything we wanted here so we could access it + // within the MCP with `this.props` + props: { userId }, + metadata: undefined + }); + + return Response.redirect(redirectTo, 302); + } + + return new Response("Method Not Allowed", { + status: 405, + headers: { allow: "GET, POST" } + }); + } +}; + +// OAuthProvider creates our worker handler +export default new OAuthProvider({ + authorizeEndpoint: "/authorize", + tokenEndpoint: "/token", + clientRegistrationEndpoint: "/register", + apiHandlers: { "/mcp": StorageMcp.serve("/mcp") }, + defaultHandler +}); +``` + +You would also add these to your `wrangler.jsonc`: + +```jsonc +{ + // rest of your config... + "r2_buckets": [{ "binding": "BUCKET", "bucket_name": "your-bucket-name" }], + "kv_namespaces": [ + { + "binding": "OAUTH_KV", // required by OAuthProvider + "id": "your-kv-id" + } + ] +} +``` + +### What's going on? + +In ~160 lines we were able to write our custom OAuth authorization flow so anyone that knows our secret password can use the MCP server. + +Just like before, in `init()` we set a few tools to access files in our R2 bucket. We also have the `whoami` tool to show users what `userId` we authenticated them with. It's just an example of how to access `props` from within the `McpAgent`. + +Most of the code here is either the HTML page to type in the password or the OAuth `/authorize` logic. +The important part is to notice how in the `OAuthProvider` we expose the `StorageMcp` through the `apiHandlers` key and use the same `serve` method we were using before. + +### Let's see how this looks like + +Once again, using https://playground.ai.cloudflare.com: +![password page](https://github.com/user-attachments/assets/8e469110-fffa-45d2-84c1-ae16a651ae41) +The auth flow prompts us for the password. + +![model calls all 3 tools after authorization](https://github.com/user-attachments/assets/07e22fef-93de-47c2-af7e-9c361e460186) +Once we've authenticated ourselves we can use all the tools! + +## Data Jurisdiction for Compliance + +`McpAgent` supports specifying a data jurisdiction for your MCP server, which is particularly useful for satisfying GDPR and other data residency regulations. By setting the `jurisdiction` option, you can ensure that your Durable Object instances (and their data) are created in a specific geographic region. + +### Using the EU Jurisdiction for GDPR + +To comply with GDPR requirements, you can specify the `"eu"` jurisdiction to ensure that all data processed by your MCP server remains within the European Union: + +```typescript +export default TinyMcp.serve("/", { + jurisdiction: "eu" +}); +``` + +Or with the OAuth-protected example: + +```typescript +export default new OAuthProvider({ + authorizeEndpoint: "/authorize", + tokenEndpoint: "/token", + clientRegistrationEndpoint: "/register", + apiHandlers: { + "/mcp": StorageMcp.serve("/mcp", { jurisdiction: "eu" }) + }, + defaultHandler +}); +``` + +When you specify `jurisdiction: "eu"`, Cloudflare will create the Durable Object instances in EU data centers, ensuring that: + +- All MCP session data stays within the EU +- User data processed by your tools remains in the EU +- State stored in the Durable Object's storage API stays in the EU + +This helps you comply with GDPR's data localization requirements without any additional configuration. + +### Available Jurisdictions + +The `jurisdiction` option accepts any value supported by [Cloudflare's Durable Objects jurisdiction API](https://developers.cloudflare.com/durable-objects/reference/data-location/), including: + +- `"eu"` - European Union +- `"fedramp"` - FedRAMP compliant locations + +## Elicitation (Human-in-the-Loop) + +MCP servers can request additional input from the user during a tool call using elicitation. This is useful for confirmation dialogs, requesting amounts, or any interactive tool flow. + +Elicitation is supported via `McpAgent` (which manages the request/response lifecycle through Durable Object storage) or via `WorkerTransport` (for stateful non-McpAgent setups). + +```typescript +import { McpAgent } from "agents/mcp"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { z } from "zod"; + +export class MyMCP extends McpAgent { + server = new McpServer({ name: "Elicitation Demo", version: "1.0.0" }); + + initialState = { counter: 0 }; + + async init() { + this.server.registerTool( + "increase-counter", + { + description: "Increase the counter", + inputSchema: { + confirm: z.boolean().describe("Do you want to increase the counter?") + } + }, + async ({ confirm }, extra) => { + if (!confirm) { + return { content: [{ type: "text", text: "Cancelled." }] }; + } + + const result = await this.server.server.elicitInput( + { + message: "By how much?", + requestedSchema: { + type: "object", + properties: { + amount: { type: "number", title: "Amount" } + }, + required: ["amount"] + } + }, + { relatedRequestId: extra.requestId } + ); + + if (result.action !== "accept" || !result.content?.amount) { + return { content: [{ type: "text", text: "Cancelled." }] }; + } + + const amount = Number(result.content.amount); + this.setState({ counter: this.state.counter + amount }); + + return { + content: [ + { + type: "text", + text: `Counter increased by ${amount}, now ${this.state.counter}` + } + ] + }; + } + ); + } +} + +export default MyMCP.serve("/mcp"); +``` + +See the [`examples/mcp-elicitation`](https://github.com/cloudflare/agents/tree/main/examples/mcp-elicitation) example for a full working demo. + +## WorkerTransport + +`WorkerTransport` is a server-side transport for running MCP servers in stateless Workers while optionally persisting session state. It is used internally by `createMcpHandler()` but can also be used directly for advanced scenarios like stateful sessions without `McpAgent`. + +```typescript +import { WorkerTransport, type TransportState } from "agents/mcp"; + +const transport = new WorkerTransport({ + sessionIdGenerator: () => crypto.randomUUID(), + enableJsonResponse: false, + storage: { + get: () => kv.get("mcp_state"), + set: (state: TransportState) => kv.put("mcp_state", state) + } +}); +``` + +Key options: + +| Option | Description | +| -------------------- | ------------------------------------------------------------------------------ | +| `sessionIdGenerator` | Function that returns a session ID for new sessions | +| `enableJsonResponse` | Return JSON instead of SSE streams (default: `false`) | +| `storage` | Optional `{ get, set }` adapter for persisting transport state across requests | +| `corsOptions` | CORS configuration | + +### Read more + +For more complex examples including authentication with third-party providers, see the [examples directory](https://github.com/cloudflare/agents/tree/main/examples). diff --git a/docs/mcp-transports.md b/docs/mcp-transports.md new file mode 100644 index 0000000000..86eebaf64f --- /dev/null +++ b/docs/mcp-transports.md @@ -0,0 +1,308 @@ +# MCP Transports + +This guide explains the different transport options for connecting to MCP servers with the Agents SDK. + +For a primer on MCP Servers and how they are implemented in the Agents SDK with `McpAgent`[here](docs/mcp-servers.md) + +## Streamable HTTP Transport (Recommended) + +The **Streamable HTTP** transport is the recommended way to connect to MCP servers. + +### How it works + +When a client connects to your MCP server: + +1. The client makes an HTTP request to your Worker with a JSON-RPC message in the body +2. Your Worker upgrades the connection to a WebSocket +3. The WebSocket connects to your `McpAgent` Durable Object which manages connection state +4. JSON-RPC messages flow bidirectionally over the WebSocket +5. Your Worker streams responses back to the client using Server-Sent Events (SSE) + +This is all handled automatically by the `McpAgent.serve()` method: + +```typescript +import { McpAgent } from "agents/mcp"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; + +export class MyMCP extends McpAgent { + server = new McpServer({ name: "Demo", version: "1.0.0" }); + + async init() { + // Define your tools, resources, prompts + } +} + +// Serve with Streamable HTTP transport +export default MyMCP.serve("/mcp"); +``` + +The `serve()` method returns a Worker with a `fetch` handler that: + +- Handles CORS preflight requests +- Manages WebSocket upgrades +- Routes messages to your Durable Object + +### Connection from clients + +Clients connect using the `streamable-http` transport: + +```typescript +await agent.addMcpServer("my-server", "https://your-worker.workers.dev/mcp"); +``` + +## Auto Transport + +The **auto** transport serves both Streamable HTTP and legacy SSE on the same endpoint. Capable clients use Streamable HTTP automatically, while older SSE-only clients continue to work. + +```typescript +export default MyMCP.serve("/mcp", { transport: "auto" }); +``` + +The handler distinguishes between the two protocols based on the request shape — no configuration or content negotiation is required from clients. This is useful when migrating from SSE to Streamable HTTP without breaking existing clients. + +## SSE Transport (Deprecated) + +We also support the legacy **SSE (Server-Sent Events)** transport, but it is deprecated in favor of Streamable HTTP. + +If you need SSE transport for compatibility: + +```typescript +// Server +export default MyMCP.serveSSE("/sse"); + +// Client +await agent.addMcpServer("my-server", url); +``` + +## RPC Transport (Experimental) + +The **RPC transport** is a custom transport designed for internal applications where your MCP server and agent are both running on Cloudflare. They can even run in the same Worker! It sends JSON-RPC messages directly over Cloudflare's RPC bindings without going over the public internet. + +### Why use RPC transport? + +- **Faster**: No network overhead - direct function calls +- **Simpler**: No HTTP endpoints, no connection management +- **Internal only**: Perfect for agents calling MCP servers within the same Worker + +**Note**: RPC transport does not support authentication. Use HTTP/SSE for external connections that require OAuth. + +### Connecting an Agent to an McpAgent via RPC + +The RPC transport uses Durable Object bindings to connect your `Agent` (MCP client) directly to your `McpAgent` (MCP server). + +#### Step 1: Define your MCP server + +Create your `McpAgent` with the tools you want to expose: + +```typescript +import { McpAgent } from "agents/mcp"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { z } from "zod"; + +type State = { counter: number }; + +export class MyMCP extends McpAgent { + server = new McpServer({ name: "MyMCP", version: "1.0.0" }); + initialState: State = { counter: 0 }; + + async init() { + this.server.tool( + "add", + "Add to the counter", + { amount: z.number() }, + async ({ amount }) => { + this.setState({ counter: this.state.counter + amount }); + return { + content: [ + { + type: "text", + text: `Added ${amount}, total is now ${this.state.counter}` + } + ] + }; + } + ); + } +} +``` + +#### Step 2: Connect your Agent to the MCP server + +In your `Agent`, call `addMcpServer()` with the Durable Object binding in `onStart()`: + +```typescript +import { AIChatAgent } from "agents/ai-chat-agent"; + +export class Chat extends AIChatAgent { + async onStart(): Promise { + // Pass the DO namespace binding directly + await this.addMcpServer("my-mcp", this.env.MyMCP); + } + + async onChatMessage(onFinish) { + const allTools = this.mcp.getAITools(); + + const result = streamText({ + model, + tools: allTools + // ... + }); + + return createUIMessageStreamResponse({ stream: result }); + } +} +``` + +RPC connections are automatically restored after Durable Object hibernation, just like HTTP connections. The binding name and props are persisted to storage so the connection can be re-established without any extra code. + +**Deduplication:** For RPC transport, if `addMcpServer` is called with a name that already has an active connection, the existing connection is returned instead of creating a duplicate. For HTTP transport, deduplication matches on both server name and URL (see [MCP Client API](./mcp-client.md) for details). This makes it safe to call `addMcpServer` in `onStart()` without worrying about creating multiple connections on restart. + +#### Step 3: Configure Durable Object bindings + +In your `wrangler.jsonc`, define bindings for both Durable Objects: + +```jsonc +{ + "durable_objects": { + "bindings": [ + { "name": "Chat", "class_name": "Chat" }, + { "name": "MyMCP", "class_name": "MyMCP" } + ] + }, + "migrations": [ + { + "new_sqlite_classes": ["MyMCP", "Chat"], + "tag": "v1" + } + ] +} +``` + +#### Step 4: Set up your Worker fetch handler + +Route requests to your Chat agent: + +```typescript +import { routeAgentRequest } from "agents"; + +export default { + async fetch(request: Request, env: Env, ctx: ExecutionContext) { + const url = new URL(request.url); + + // Optionally expose the MCP server via HTTP as well + if (url.pathname.startsWith("/mcp")) { + return MyMCP.serve("/mcp").fetch(request, env, ctx); + } + + const response = await routeAgentRequest(request, env); + if (response) return response; + + return new Response("Not found", { status: 404 }); + } +}; +``` + +### Passing props from client to server + +Since RPC transport does not have an OAuth flow, you can pass user context (like userId, role, etc.) directly as props: + +```typescript +await this.addMcpServer("my-mcp", this.env.MyMCP, { + props: { userId: "user-123", role: "admin" } +}); +``` + +Your `McpAgent` can then access these props: + +```typescript +export class MyMCP extends McpAgent< + Env, + State, + { userId?: string; role?: string } +> { + async init() { + this.server.tool("whoami", "Get current user info", {}, async () => { + const userId = this.props?.userId || "anonymous"; + const role = this.props?.role || "guest"; + + return { + content: [{ type: "text", text: `User ID: ${userId}, Role: ${role}` }] + }; + }); + } +} +``` + +The props are: + +- **Type-safe**: TypeScript extracts the Props type from your McpAgent generic +- **Persistent**: Stored in Durable Object storage via `updateProps()` +- **Available immediately**: Set before any tool calls are made + +This is useful for: + +- User authentication context +- Tenant/organization IDs +- Feature flags or permissions +- Any per-connection configuration + +### How RPC transport works under the hood + +When you call `addMcpServer()` with a Durable Object binding, the SDK: + +1. Creates an `RPCClientTransport` that wraps the DO stub +2. Calls `handleMcpMessage()` on the `McpAgent` for each JSON-RPC message +3. The `McpAgent` routes messages through its `RPCServerTransport` to the MCP server +4. Responses flow back synchronously through the RPC call + +This happens entirely within your Worker's execution context using Cloudflare's RPC mechanism - no HTTP, no WebSockets, no public internet. + +The RPC transport fully supports: + +- JSON-RPC 2.0 validation (via the MCP SDK's schema) +- Batch requests +- Notifications (messages without `id` field) +- Automatic reconnection after Durable Object hibernation (when called from `onStart()`) + +### Configuring RPC Transport Server Timeout + +The RPC transport has a configurable timeout for waiting for tool responses. By default, the server will wait **60 seconds** for a tool handler to respond. You can customize this by overriding `getRpcTransportOptions()` in your `McpAgent`: + +```typescript +export class MyMCP extends McpAgent { + server = new McpServer({ name: "MyMCP", version: "1.0.0" }); + + protected getRpcTransportOptions() { + return { timeout: 120000 }; // 2 minutes + } + + async init() { + this.server.tool( + "long-running-task", + "A tool that takes a while", + { input: z.string() }, + async ({ input }) => { + await longRunningOperation(input); + return { + content: [{ type: "text", text: "Task completed" }] + }; + } + ); + } +} +``` + +## Choosing a transport + +| Transport | Use when | Pros | Cons | +| ------------------- | ---------------------------------------- | ---------------------------------------- | ------------------------------- | +| **Streamable HTTP** | External MCP servers, production apps | Standard protocol, secure, supports auth | Slight network overhead | +| **Auto** | Migrating from SSE, mixed client support | Serves both protocols on one endpoint | Reserves `{path}/message` route | +| **RPC** | Internal agents | Fastest, simplest setup | No auth, Service Bindings only | +| **SSE** | Legacy compatibility | Backwards compatible | Deprecated, use Streamable HTTP | + +## Examples + +- **Streamable HTTP**: See [`examples/mcp`](../examples/mcp) +- **RPC Transport**: See [`examples/mcp-rpc-transport`](../examples/mcp-rpc-transport) +- **MCP Client**: See [`examples/mcp-client`](../examples/mcp-client) diff --git a/docs/migration-to-ai-sdk-v5.md b/docs/migration-to-ai-sdk-v5.md new file mode 100644 index 0000000000..ef05baf2e4 --- /dev/null +++ b/docs/migration-to-ai-sdk-v5.md @@ -0,0 +1,96 @@ +# Migrating from AI SDK v4 to v5 + +This guide covers the changes needed when upgrading from AI SDK v4 to v5 with `@cloudflare/ai-chat`. + +> If you are on AI SDK v5 and upgrading to v6, see the [v6 migration guide](./migration-to-ai-sdk-v6.md) instead. + +## Message format: `content` to `parts` + +The biggest change. Messages now use a `parts` array instead of a `content` string: + +```typescript +// v4 +const message = { id: "1", role: "user", content: "Hello" }; + +// v5 +const message = { + id: "1", + role: "user", + parts: [{ type: "text", text: "Hello" }] +}; +``` + +**You do not need to migrate stored messages manually.** `AIChatAgent` automatically transforms legacy messages on load via `autoTransformMessages()`. This handles v4 `content` strings, tool invocations, reasoning parts, file data, and malformed formats. + +## Import changes + +```typescript +// v4 +import type { Message } from "ai"; +import { useChat } from "ai/react"; + +// v5 +import type { UIMessage } from "ai"; +import { useChat } from "@ai-sdk/react"; +``` + +## Tool definitions: `parameters` to `inputSchema` + +```typescript +// v4 +const tools = { + weather: { + description: "Get weather", + parameters: z.object({ city: z.string() }), + execute: async ({ city }) => fetchWeather(city) + } +}; + +// v5 +const tools = { + weather: { + description: "Get weather", + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }) => fetchWeather(city) + } +}; +``` + +## Streaming events + +v5 adds `text-start` and `text-end` events around text deltas, and renames `textDelta` to `delta`: + +```typescript +// v4 +chunk.type === "text-delta" && chunk.textDelta; + +// v5 +chunk.type === "text-delta" && chunk.delta; +// Plus new: "text-start" and "text-end" events +``` + +## Migration checklist + +1. Update dependencies: `npm update agents ai` +2. Replace `import type { Message }` with `import type { UIMessage }` +3. Replace `"ai/react"` imports with `"@ai-sdk/react"` +4. Rename `parameters` to `inputSchema` in tool definitions +5. Run `npm run typecheck` and fix any remaining type errors +6. Test your application -- legacy stored messages are migrated automatically + +## Migration utilities (deprecated) + +These are available but rarely needed since migration is automatic: + +```typescript +import { + autoTransformMessages, // Used internally by AIChatAgent + migrateMessagesToUIFormat, // Deprecated -- use autoTransformMessages + analyzeCorruption // Deprecated -- debugging only +} from "@cloudflare/ai-chat/ai-chat-v5-migration"; +``` + +## Further reading + +- [Official AI SDK v5 migration guide](https://ai-sdk.dev/docs/migration-guides/migration-guide-5-0) +- [v6 migration guide](./migration-to-ai-sdk-v6.md) (if upgrading further) diff --git a/docs/migration-to-ai-sdk-v6.md b/docs/migration-to-ai-sdk-v6.md new file mode 100644 index 0000000000..f892d05430 --- /dev/null +++ b/docs/migration-to-ai-sdk-v6.md @@ -0,0 +1,163 @@ +# Migrating from AI SDK v5 to v6 + +This guide covers the changes needed when upgrading from AI SDK v5 to v6 with `@cloudflare/ai-chat`. + +## Installation + +```bash +npm install ai@latest @ai-sdk/react@latest @ai-sdk/openai@latest +``` + +## Breaking changes + +### 1. `convertToModelMessages()` is now async + +Add `await` to all calls: + +```typescript +// v5 +const result = streamText({ + messages: convertToModelMessages(this.messages), + model: openai("gpt-4o") +}); + +// v6 +const result = streamText({ + messages: await convertToModelMessages(this.messages), + model: openai("gpt-4o") +}); +``` + +### 2. `CoreMessage` removed + +Replace `CoreMessage` with `ModelMessage` and `convertToCoreMessages()` with `convertToModelMessages()`: + +```typescript +// v5 +import { convertToCoreMessages, type CoreMessage } from "ai"; + +// v6 +import { convertToModelMessages, type ModelMessage } from "ai"; +``` + +### 3. Tool pattern: server-side tools (recommended) + +v6 introduces `needsApproval` and the `onToolCall` callback. For most apps, define tools on the server with `tool()` from `"ai"` for full Zod type safety: + +**Before (v5):** + +```typescript +// Client defined tools with AITool type +useAgentChat({ + agent, + tools: clientTools, + experimental_automaticToolResolution: true, + toolsRequiringConfirmation: ["askConfirmation"] +}); +``` + +**After (v6):** + +```typescript +// Server: all tools defined here +const tools = { + getWeather: tool({ + description: "Get weather", + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }) => fetchWeather(city) + }), + getLocation: tool({ + description: "Get user location", + inputSchema: z.object({}) + // No execute -- client handles via onToolCall + }), + processPayment: tool({ + description: "Process payment", + inputSchema: z.object({ amount: z.number() }), + needsApproval: async ({ amount }) => amount > 100, + execute: async ({ amount }) => charge(amount) + }) +}; + +// Client: handle tools via callbacks +useAgentChat({ + agent, + onToolCall: async ({ toolCall, addToolOutput }) => { + if (toolCall.toolName === "getLocation") { + const pos = await getPosition(); + addToolOutput({ + toolCallId: toolCall.toolCallId, + output: { lat: pos.coords.latitude, lng: pos.coords.longitude } + }); + } + } +}); +``` + +**Dynamic client tools (SDK/platform pattern):** + +If you are building an SDK or platform where tools are defined dynamically by the embedding application at runtime, the `tools` option on `useAgentChat` and `createToolsFromClientSchemas()` on the server are still fully supported: + +```typescript +// Server: accept whatever tools the client sends +const tools = { + ...createToolsFromClientSchemas(options.clientTools), + ...serverTools +}; + +// Client: register tools dynamically +useAgentChat({ + agent, + tools: dynamicTools, + onToolCall: async ({ toolCall, addToolOutput }) => { + const tool = dynamicTools[toolCall.toolName]; + if (tool?.execute) { + const output = await tool.execute(toolCall.input); + addToolOutput({ toolCallId: toolCall.toolCallId, output }); + } + } +}); +``` + +### 4. `generateObject` mode option removed + +Remove `mode: "json"` or similar from `generateObject` calls. + +### 5. `isToolUIPart` and `getToolName` now include dynamic tools + +In v6, these check both static and dynamic tool parts. For the old behavior, use `isStaticToolUIPart` and `getStaticToolName`. Most users do not need to change anything. + +## Deprecated APIs + +| Deprecated | Replacement | +| -------------------------------------- | --------------------------------------------------------- | +| `toolsRequiringConfirmation` | [`needsApproval`](./human-in-the-loop.md) on server tools | +| `experimental_automaticToolResolution` | [`onToolCall`](./client-tools-continuation.md) callback | +| `addToolResult()` | `addToolOutput()` or `addToolApprovalResponse()` | + +**Not deprecated:** `AITool`, `createToolsFromClientSchemas()`, `extractClientToolSchemas()`, and the `tools` option on `useAgentChat` are supported for SDK/platform use cases where tools are defined dynamically at runtime. + +## Migration checklist + +**Packages:** + +- `ai` to `^6.0.0` +- `@ai-sdk/react` to `^3.0.0` +- `@ai-sdk/openai` (and other providers) to `^3.0.0` + +**Code changes:** + +- Add `await` to all `convertToModelMessages()` calls +- Replace `CoreMessage` with `ModelMessage` +- Replace `convertToCoreMessages()` with `convertToModelMessages()` +- Remove `mode` from `generateObject` calls +- Move static tool definitions to server using `tool()` (recommended for most apps) +- Use `onToolCall` in `useAgentChat` for client-side tool execution +- Replace `toolsRequiringConfirmation` with `needsApproval` +- Replace `addToolResult()` with `addToolOutput()` or `addToolApprovalResponse()` + +## Further reading + +- [Official AI SDK v6 migration guide](https://ai-sdk.dev/docs/migration-guides/migration-guide-6-0) +- [Human in the Loop](./human-in-the-loop.md) -- `needsApproval` and `addToolApprovalResponse` +- [Client Tools](./client-tools-continuation.md) -- `onToolCall` and auto-continuation diff --git a/docs/observability.md b/docs/observability.md new file mode 100644 index 0000000000..d0bdd5043d --- /dev/null +++ b/docs/observability.md @@ -0,0 +1,203 @@ +# Observability + +Agents emit structured events for every significant operation — RPC calls, state changes, schedule execution, workflow transitions, MCP connections, and more. These events are published to [diagnostics channels](https://developers.cloudflare.com/workers/runtime-apis/nodejs/diagnostics-channel/) and are silent by default (zero overhead when nobody is listening). + +## Event structure + +Every event has these fields: + +```ts +{ + type: "rpc", // what happened + agent: "MyAgent", // which agent class emitted it + name: "user-123", // which agent instance (Durable Object name) + payload: { method: "getWeather" }, // details + timestamp: 1758005142787 // when (ms since epoch) +} +``` + +`agent` and `name` identify the source agent — `agent` is the class name and `name` is the Durable Object instance name. + +## Channels + +Events are routed to eight named channels based on their type: + +| Channel | Event types | Description | +| ------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------- | +| `agents:state` | `state:update` | State sync events | +| `agents:rpc` | `rpc`, `rpc:error` | RPC method calls and failures | +| `agents:message` | `message:request`, `message:response`, `message:clear`, `message:cancel`, `message:error`, `tool:result`, `tool:approval` | Chat message and tool lifecycle | +| `agents:schedule` | `schedule:create`, `schedule:execute`, `schedule:cancel`, `schedule:retry`, `schedule:error`, `schedule:duplicate_warning`, `queue:create`, `queue:retry`, `queue:error` | Scheduled and queued task lifecycle | +| `agents:lifecycle` | `connect`, `disconnect`, `destroy` | Agent connection and teardown | +| `agents:workflow` | `workflow:start`, `workflow:event`, `workflow:approved`, `workflow:rejected`, `workflow:terminated`, `workflow:paused`, `workflow:resumed`, `workflow:restarted` | Workflow state transitions | +| `agents:mcp` | `mcp:client:preconnect`, `mcp:client:connect`, `mcp:client:authorize`, `mcp:client:discover`, `mcp:client:close` | MCP client operations | +| `agents:email` | `email:receive`, `email:reply` | Email processing | + +## Subscribing to events + +### Typed subscribe helper + +The `subscribe()` function from `agents/observability` provides type-safe access to events on a specific channel: + +```ts +import { subscribe } from "agents/observability"; + +const unsub = subscribe("rpc", (event) => { + if (event.type === "rpc") { + console.log(`RPC call: ${event.payload.method}`); + } + if (event.type === "rpc:error") { + console.error( + `RPC failed: ${event.payload.method} — ${event.payload.error}` + ); + } +}); + +// Clean up when done +unsub(); +``` + +The callback is fully typed — `event` is narrowed to only the event types that flow through that channel. + +### Raw diagnostics_channel + +You can also subscribe directly using the Node.js API: + +```ts +import { subscribe } from "node:diagnostics_channel"; + +subscribe("agents:schedule", (event) => { + console.log(event); +}); +``` + +## Tail Workers (production) + +In production, all diagnostics channel messages are automatically forwarded to [Tail Workers](https://developers.cloudflare.com/workers/observability/tail-workers/). No subscription code is needed in the agent itself — attach a Tail Worker and access events via `event.diagnosticsChannelEvents`: + +```ts +export default { + async tail(events) { + for (const event of events) { + for (const msg of event.diagnosticsChannelEvents) { + // msg.channel is "agents:rpc", "agents:workflow", etc. + // msg.message is the typed event payload + console.log(msg.timestamp, msg.channel, msg.message); + } + } + } +}; +``` + +This gives you structured, filterable observability in production with zero overhead in the agent hot path. + +## Custom observability + +You can override the default implementation by providing your own `Observability` interface: + +```ts +import { Agent } from "agents"; +import type { Observability } from "agents/observability"; + +const myObservability: Observability = { + emit(event) { + // Send to your logging service, filter events, etc. + if (event.type === "rpc:error") { + myLogger.error(event.payload.method, event.payload.error); + } + } +}; + +class MyAgent extends Agent { + override observability = myObservability; +} +``` + +Set `observability` to `undefined` to disable all event emission: + +```ts +class MyAgent extends Agent { + override observability = undefined; +} +``` + +## Event reference + +### RPC events + +| Type | Payload | When | +| ----------- | ------------------------ | ------------------------------- | +| `rpc` | `{ method, streaming? }` | A `@callable` method is invoked | +| `rpc:error` | `{ method, error }` | A `@callable` method throws | + +### State events + +| Type | Payload | When | +| -------------- | ------- | ---------------------- | +| `state:update` | `{}` | `setState()` is called | + +### Message and tool events (`AIChatAgent`) + +These events are emitted by `AIChatAgent` from `@cloudflare/ai-chat`. They track the chat message lifecycle, including client-side tool interactions. + +| Type | Payload | When | +| ------------------ | -------------------------- | ----------------------------------- | +| `message:request` | `{}` | A chat message is received | +| `message:response` | `{}` | A chat response stream completes | +| `message:clear` | `{}` | Chat history is cleared | +| `message:cancel` | `{ requestId }` | A streaming request is cancelled | +| `message:error` | `{ error }` | A chat stream fails | +| `tool:result` | `{ toolCallId, toolName }` | A client tool result is received | +| `tool:approval` | `{ toolCallId, approved }` | A tool call is approved or rejected | + +### Schedule and queue events + +| Type | Payload | When | +| ---------------------------- | ---------------------------------------- | -------------------------------------------- | +| `schedule:create` | `{ callback, id }` | A schedule is created | +| `schedule:execute` | `{ callback, id }` | A scheduled callback starts | +| `schedule:cancel` | `{ callback, id }` | A schedule is cancelled | +| `schedule:retry` | `{ callback, id, attempt, maxAttempts }` | A scheduled callback is retried | +| `schedule:error` | `{ callback, id, error, attempts }` | A scheduled callback fails after all retries | +| `schedule:duplicate_warning` | `{ callback, count, type }` | Duplicate schedules detected for a callback | +| `queue:create` | `{ callback, id }` | A task is enqueued | +| `queue:retry` | `{ callback, id, attempt, maxAttempts }` | A queued callback is retried | +| `queue:error` | `{ callback, id, error, attempts }` | A queued callback fails after all retries | + +### Lifecycle events + +| Type | Payload | When | +| ------------ | -------------------------------- | ------------------------------------- | +| `connect` | `{ connectionId }` | A WebSocket connection is established | +| `disconnect` | `{ connectionId, code, reason }` | A WebSocket connection is closed | +| `destroy` | `{}` | The agent is destroyed | + +### Workflow events + +| Type | Payload | When | +| --------------------- | ------------------------------- | ------------------------------ | +| `workflow:start` | `{ workflowId, workflowName? }` | A workflow instance is started | +| `workflow:event` | `{ workflowId, eventType? }` | An event is sent to a workflow | +| `workflow:approved` | `{ workflowId, reason? }` | A workflow is approved | +| `workflow:rejected` | `{ workflowId, reason? }` | A workflow is rejected | +| `workflow:terminated` | `{ workflowId, workflowName? }` | A workflow is terminated | +| `workflow:paused` | `{ workflowId, workflowName? }` | A workflow is paused | +| `workflow:resumed` | `{ workflowId, workflowName? }` | A workflow is resumed | +| `workflow:restarted` | `{ workflowId, workflowName? }` | A workflow is restarted | + +### MCP events + +| Type | Payload | When | +| ----------------------- | -------------------------------------------- | ---------------------------------------------------------------------------------- | +| `mcp:client:preconnect` | `{ serverId }` | Before connecting to an MCP server | +| `mcp:client:connect` | `{ url, transport, state, error? }` | An MCP connection attempt completes or fails | +| `mcp:client:authorize` | `{ serverId, authUrl, clientId? }` | An MCP OAuth flow begins | +| `mcp:client:discover` | `{ url?, state?, error?, capability? }` | MCP capability discovery succeeds or fails | +| `mcp:client:close` | `{ url, transport?, state, error?, phase? }` | An MCP connection is closed (`phase` is `"terminate-session"` or `"client-close"`) | + +### Email events + +| Type | Payload | When | +| --------------- | ------------------------ | --------------------- | +| `email:receive` | `{ from, to, subject? }` | An email is received | +| `email:reply` | `{ from, to, subject? }` | A reply email is sent | diff --git a/docs/push-notifications.md b/docs/push-notifications.md new file mode 100644 index 0000000000..b7f7e27647 --- /dev/null +++ b/docs/push-notifications.md @@ -0,0 +1,367 @@ +# Push Notifications + +Send browser push notifications from your agent — even when the user has closed the tab. By combining the agent's persistent state (for storing push subscriptions), scheduling (for timed delivery), and the [Web Push API](https://developer.mozilla.org/en-US/docs/Web/API/Push_API), you can reach users who are completely offline. + +## How It Works + +``` +Browser Agent (Durable Object) +─────── ────────────────────── +1. Register service worker +2. Subscribe to push (VAPID key) +3. Send subscription to agent ──────► Store in this.state +4. Create reminder ─────────────────► this.schedule(delay, "sendReminder", payload) + + ... user closes tab ... + +5. Alarm fires → sendReminder() + web-push sends encrypted payload + │ +6. Service worker receives push ◄─────────────┘ +7. showNotification() +``` + +The agent stores push subscriptions durably in its state and uses `this.schedule()` to fire notifications at the right time. When the alarm triggers, the agent calls the push service endpoint using the [`web-push`](https://www.npmjs.com/package/web-push) library. The browser's service worker receives the push event and displays a native notification. + +## Prerequisites + +### Generate VAPID Keys + +Web Push requires a VAPID (Voluntary Application Server Identification) key pair. Generate one: + +```bash +npx web-push generate-vapid-keys +``` + +Store the keys in a `.env` file for local development: + +``` +VAPID_PUBLIC_KEY=BGxK... +VAPID_PRIVATE_KEY=abc1... +VAPID_SUBJECT=mailto:you@example.com +``` + +For production, use `wrangler secret put`: + +```bash +wrangler secret put VAPID_PUBLIC_KEY +wrangler secret put VAPID_PRIVATE_KEY +wrangler secret put VAPID_SUBJECT +``` + +## Create the Agent + +The agent has three responsibilities: store push subscriptions, schedule reminders, and send notifications when alarms fire. + +```typescript +import { Agent, callable, routeAgentRequest } from "agents"; +import webpush from "web-push"; + +type Subscription = { + endpoint: string; + expirationTime: number | null; + keys: { + p256dh: string; + auth: string; + }; +}; + +type Reminder = { + id: string; + message: string; + scheduledAt: number; + sent: boolean; +}; + +type ReminderAgentState = { + subscriptions: Subscription[]; + reminders: Reminder[]; +}; + +export class ReminderAgent extends Agent { + initialState: ReminderAgentState = { + subscriptions: [], + reminders: [] + }; + + @callable() + getVapidPublicKey(): string { + return this.env.VAPID_PUBLIC_KEY; + } + + @callable() + async subscribe(subscription: Subscription): Promise<{ ok: boolean }> { + const exists = this.state.subscriptions.some( + (s) => s.endpoint === subscription.endpoint + ); + if (!exists) { + this.setState({ + ...this.state, + subscriptions: [...this.state.subscriptions, subscription] + }); + } + return { ok: true }; + } + + @callable() + async unsubscribe(endpoint: string): Promise<{ ok: boolean }> { + this.setState({ + ...this.state, + subscriptions: this.state.subscriptions.filter( + (s) => s.endpoint !== endpoint + ) + }); + return { ok: true }; + } + + @callable() + async createReminder( + message: string, + delaySeconds: number + ): Promise { + const id = crypto.randomUUID(); + const scheduledAt = Date.now() + delaySeconds * 1000; + const reminder: Reminder = { id, message, scheduledAt, sent: false }; + + this.setState({ + ...this.state, + reminders: [...this.state.reminders, reminder] + }); + + await this.schedule(delaySeconds, "sendReminder", { id, message }); + + return reminder; + } +``` + +When the scheduled alarm fires, send the push notification to all stored subscriptions: + +```typescript + async sendReminder(payload: { id: string; message: string }) { + webpush.setVapidDetails( + this.env.VAPID_SUBJECT, + this.env.VAPID_PUBLIC_KEY, + this.env.VAPID_PRIVATE_KEY + ); + + const deadEndpoints: string[] = []; + + await Promise.all( + this.state.subscriptions.map(async (sub) => { + try { + await webpush.sendNotification( + sub, + JSON.stringify({ + title: "Reminder", + body: payload.message, + tag: `reminder-${payload.id}` + }) + ); + } catch (err: unknown) { + const statusCode = + err instanceof webpush.WebPushError ? err.statusCode : 0; + if (statusCode === 404 || statusCode === 410) { + deadEndpoints.push(sub.endpoint); + } + } + }) + ); + + // Clean up expired or revoked subscriptions + if (deadEndpoints.length > 0) { + this.setState({ + ...this.state, + subscriptions: this.state.subscriptions.filter( + (s) => !deadEndpoints.includes(s.endpoint) + ) + }); + } + + // Mark reminder as sent + this.setState({ + ...this.state, + reminders: this.state.reminders.map((r) => + r.id === payload.id ? { ...r, sent: true } : r + ) + }); + + // Notify any connected clients in real time + this.broadcast( + JSON.stringify({ + type: "reminder_sent", + id: payload.id, + timestamp: Date.now() + }) + ); + } +} +``` + +The `sendReminder` callback handles three things: delivering the push notification via the `web-push` library, cleaning up dead subscriptions (the push service returns 404 or 410 when a subscription is no longer valid), and broadcasting to any connected clients so the UI updates in real time. + +## Set Up the Service Worker + +The service worker runs in the browser and receives push events even when no tabs are open. Place this file at `public/sw.js` so it is served from the root of your domain: + +```javascript +self.addEventListener("push", (event) => { + if (!event.data) return; + + const data = event.data.json(); + + event.waitUntil( + self.registration.showNotification(data.title || "Notification", { + body: data.body || "", + icon: data.icon || "/favicon.ico", + tag: data.tag, + data: data.data + }) + ); +}); + +self.addEventListener("notificationclick", (event) => { + event.notification.close(); + + event.waitUntil( + self.clients.matchAll({ type: "window" }).then((windowClients) => { + for (const client of windowClients) { + if (client.url.includes(self.location.origin) && "focus" in client) { + return client.focus(); + } + } + return self.clients.openWindow("/"); + }) + ); +}); +``` + +The `push` event handler parses the JSON payload and displays a native notification. The `notificationclick` handler focuses an existing tab or opens a new one when the user taps the notification. + +## Build the Client + +The client needs to: register the service worker, request notification permission, subscribe to push using the VAPID public key, and send the subscription to the agent. + +### Register the Service Worker + +```typescript +useEffect(() => { + if (!("serviceWorker" in navigator) || !("PushManager" in window)) { + return; // Push not supported + } + + navigator.serviceWorker.register("/sw.js"); +}, []); +``` + +### Subscribe to Push + +Fetch the VAPID public key from the agent, then subscribe through the Push API: + +```typescript +function base64urlToUint8Array(base64url: string): Uint8Array { + const padded = base64url + "=".repeat((4 - (base64url.length % 4)) % 4); + const binary = atob(padded.replace(/-/g, "+").replace(/_/g, "/")); + const bytes = new Uint8Array(binary.length); + for (let i = 0; i < binary.length; i++) bytes[i] = binary.charCodeAt(i); + return bytes; +} + +async function subscribeToPush(agent) { + const permission = await Notification.requestPermission(); + if (permission !== "granted") return; + + const vapidPublicKey = await agent.call("getVapidPublicKey"); + const reg = await navigator.serviceWorker.ready; + const subscription = await reg.pushManager.subscribe({ + userVisibleOnly: true, + applicationServerKey: base64urlToUint8Array(vapidPublicKey).buffer + }); + + const subJson = subscription.toJSON(); + await agent.call("subscribe", [ + { + endpoint: subJson.endpoint, + expirationTime: subJson.expirationTime ?? null, + keys: subJson.keys + } + ]); +} +``` + +### Create Reminders + +With the subscription stored, creating a reminder is a single RPC call. The agent handles scheduling and delivery: + +```typescript +await agent.call("createReminder", ["Check the oven", 300]); +``` + +The agent schedules an alarm for 300 seconds (5 minutes). When it fires, the push notification arrives — even if the user closed the tab minutes ago. + +## Configuration + +### wrangler.jsonc + +```jsonc +{ + "name": "push-notifications", + "compatibility_date": "2026-01-28", + "compatibility_flags": ["nodejs_compat"], + "main": "src/server.ts", + "durable_objects": { + "bindings": [{ "name": "ReminderAgent", "class_name": "ReminderAgent" }] + }, + "migrations": [{ "tag": "v1", "new_sqlite_classes": ["ReminderAgent"] }], + "assets": { + "not_found_handling": "single-page-application" + } +} +``` + +The `nodejs_compat` compatibility flag is required for the `web-push` library. + +### Dependencies + +```bash +npm install agents web-push +``` + +## Production Considerations + +### Subscription Expiry + +Push subscriptions can expire or be revoked by the user. Always handle 404 and 410 responses from the push service by removing the dead subscription from state, as shown in the `sendReminder` example above. + +### Per-User vs Shared Agents + +For most applications, use one agent per user (using the user ID as the agent name). This isolates each user's subscriptions and reminders. For broadcast-style notifications (same message to many users), a shared agent can store all subscriptions, but be aware of the state size as the subscription list grows. + +### Combining Push with WebSocket Broadcast + +Use `this.broadcast()` for clients that are currently connected (instant, no push service roundtrip) and Web Push for clients that are offline. The `sendReminder` example above does both — connected clients get a real-time WebSocket message, and offline clients get a push notification. + +### Multiple Devices + +A single user may subscribe from multiple browsers or devices. The agent stores each subscription separately, and `sendReminder` iterates over all of them. Each device receives its own push notification. + +### Retry on Failure + +If the push service returns a 5xx error (temporary failure), you can retry using `this.schedule()` with a short delay: + +```typescript +try { + await webpush.sendNotification(sub, payload); +} catch (err: unknown) { + const statusCode = err instanceof webpush.WebPushError ? err.statusCode : 0; + if (statusCode >= 500) { + await this.schedule(60, "retrySendNotification", { + endpoint: sub.endpoint, + payload + }); + } +} +``` + +## Full Example + +See the complete working example at [`examples/push-notifications/`](../examples/push-notifications/) — includes the agent, service worker, and a React client with subscription management, reminder creation, and real-time state sync. diff --git a/docs/queue.md b/docs/queue.md new file mode 100644 index 0000000000..1d4a615b1b --- /dev/null +++ b/docs/queue.md @@ -0,0 +1,329 @@ +# Queue System + +The Agents SDK provides a built-in queue system that allows you to schedule tasks for asynchronous execution. This is particularly useful for background processing, delayed operations, and managing workloads that don't need immediate execution. + +## Overview + +The queue system is built into the base `Agent` class. Tasks are stored in a SQLite table and processed automatically in FIFO (First In, First Out) order. + +## QueueItem Type + +```typescript +export type QueueItem = { + id: string; // Unique identifier for the queued task + payload: T; // Data to pass to the callback function + callback: keyof Agent; // Name of the method to call + created_at: number; // Timestamp when the task was created + retry?: RetryOptions; // Retry options (if configured) +}; +``` + +## Core Methods + +### queue() + +Adds a task to the queue for future execution. + +```typescript +async queue( + callback: keyof this, + payload: T, + options?: { retry?: RetryOptions } +): Promise +``` + +**Parameters:** + +- `callback`: The name of the method to call when processing the task +- `payload`: Data to pass to the callback method +- `options.retry`: Optional retry configuration. See [Retries](./retries.md) for details. + +**Returns:** The unique ID of the queued task + +**Example:** + +```typescript +class MyAgent extends Agent { + async processEmail(data: { email: string; subject: string }) { + // Process the email + console.log(`Processing email: ${data.subject}`); + } + + async onMessage(message: string) { + // Queue an email processing task + const taskId = await this.queue("processEmail", { + email: "user@example.com", + subject: "Welcome!" + }); + + console.log(`Queued task with ID: ${taskId}`); + } +} +``` + +### dequeue() + +Removes a specific task from the queue by ID. + +```typescript +dequeue(id: string): void +``` + +**Parameters:** + +- `id`: The ID of the task to remove + +**Example:** + +```typescript +// Remove a specific task +agent.dequeue("abc123def"); +``` + +### dequeueAll() + +Removes all tasks from the queue. + +```typescript +dequeueAll(): void +``` + +**Example:** + +```typescript +// Clear the entire queue +agent.dequeueAll(); +``` + +### dequeueAllByCallback() + +Removes all tasks that match a specific callback method. + +```typescript +dequeueAllByCallback(callback: string): void +``` + +**Parameters:** + +- `callback`: Name of the callback method + +**Example:** + +```typescript +// Remove all email processing tasks +agent.dequeueAllByCallback("processEmail"); +``` + +### getQueue() + +Retrieves a specific queued task by ID. + +```typescript +getQueue(id: string): QueueItem | undefined +``` + +**Parameters:** + +- `id`: The ID of the task to retrieve + +**Returns:** The QueueItem with parsed payload or undefined if not found + +**Note:** The payload is automatically parsed from JSON before being returned + +**Example:** + +```typescript +const task = agent.getQueue("abc123def"); +if (task) { + console.log(`Task callback: ${task.callback}`); + console.log(`Task payload:`, task.payload); +} +``` + +### getQueues() + +Retrieves all queued tasks that match a specific key-value pair in their payload. + +```typescript +getQueues(key: string, value: string): QueueItem[] +``` + +**Parameters:** + +- `key`: The key to filter by in the payload +- `value`: The value to match + +**Returns:** Array of matching QueueItem objects + +**Note:** This method fetches all queue items and filters them in memory by parsing each payload and checking if the specified key matches the value + +**Example:** + +```typescript +// Find all tasks for a specific user +const userTasks = await agent.getQueues("userId", "12345"); +``` + +## How Queue Processing Works + +1. **Validation**: When calling `queue()`, the method validates that the callback exists as a function on the agent +2. **Automatic Processing**: After queuing, the system automatically attempts to flush the queue +3. **FIFO Order**: Tasks are processed in the order they were created (`created_at` timestamp) +4. **Context Preservation**: Each queued task runs with the same agent context (connection, request, email) +5. **Automatic Retries**: If a callback fails, it is retried with exponential backoff (configurable per task) +6. **Automatic Dequeue**: Tasks are removed from the queue after successful execution or after all retry attempts are exhausted +7. **Error Handling**: If a callback method does not exist at execution time, an error is logged and the task is skipped +8. **Persistence**: Tasks are stored in the `cf_agents_queues` table and survive agent restarts + +## Queue Callback Methods + +When defining callback methods for queued tasks, they must follow this signature: + +```typescript +async callbackMethod(payload: unknown, queueItem: QueueItem): Promise +``` + +**Example:** + +```typescript +class MyAgent extends Agent { + async sendNotification( + payload: { userId: string; message: string }, + queueItem: QueueItem + ) { + console.log(`Processing task ${queueItem.id}`); + console.log( + `Sending notification to user ${payload.userId}: ${payload.message}` + ); + + // Your notification logic here + await this.notificationService.send(payload.userId, payload.message); + } + + async onUserSignup(userData: any) { + // Queue a welcome notification + await this.queue("sendNotification", { + userId: userData.id, + message: "Welcome to our platform!" + }); + } +} +``` + +## Use Cases + +### Background Processing + +```typescript +class DataProcessor extends Agent { + async processLargeDataset(data: { datasetId: string; userId: string }) { + const results = await this.heavyComputation(data.datasetId); + await this.notifyUser(data.userId, results); + } + + async onDataUpload(uploadData: any) { + // Queue the processing instead of doing it synchronously + await this.queue("processLargeDataset", { + datasetId: uploadData.id, + userId: uploadData.userId + }); + + return { message: "Data upload received, processing started" }; + } +} +``` + +### Delayed Operations + +```typescript +class ReminderAgent extends Agent { + async sendReminder(data: { userId: string; message: string }) { + await this.emailService.send(data.userId, data.message); + } + + async scheduleReminder(userId: string, message: string, delayMs: number) { + // Note: For true delayed execution, combine with the scheduling system + // This example shows queueing for later processing + await this.queue("sendReminder", { userId, message }); + } +} +``` + +### Batch Operations + +```typescript +class BatchProcessor extends Agent { + async processBatch(data: { items: any[]; batchId: string }) { + for (const item of data.items) { + await this.processItem(item); + } + console.log(`Completed batch ${data.batchId}`); + } + + async onLargeRequest(items: any[]) { + // Split large requests into smaller batches + const batchSize = 10; + for (let i = 0; i < items.length; i += batchSize) { + const batch = items.slice(i, i + batchSize); + await this.queue("processBatch", { + items: batch, + batchId: `batch-${i / batchSize + 1}` + }); + } + } +} +``` + +## Best Practices + +1. **Keep Payloads Small**: Payloads are JSON-serialized and stored in the database +2. **Idempotent Operations**: Design callback methods to be safe to retry +3. **Error Handling**: Include proper error handling in callback methods +4. **Monitoring**: Use logging to track queue processing +5. **Cleanup**: Regularly clean up completed or failed tasks if needed + +## Error Handling and Retries + +Queued tasks are automatically retried on failure with exponential backoff. The default is 3 attempts. You can customize this per task: + +```typescript +// Retry up to 5 times with custom backoff +await this.queue("reliableTask", payload, { + retry: { maxAttempts: 5, baseDelayMs: 500 } +}); +``` + +If you need custom error handling in the callback: + +```typescript +class RobustAgent extends Agent { + async reliableTask(payload: { data: string }, queueItem: QueueItem) { + try { + await this.doSomethingRisky(payload); + } catch (error) { + console.error(`Task ${queueItem.id} failed:`, error); + // The retry system will catch this error and retry automatically + throw error; + } + } +} +``` + +See [Retries](./retries.md) for full documentation on retry options and patterns. + +## Integration with Other Features + +The queue system works seamlessly with other Agent SDK features: + +- **State Management**: Access agent state within queued callbacks +- **Scheduling**: Combine with `schedule()` for time-based queue processing +- **Retries**: Built-in retry with exponential backoff. See [Retries](./retries.md). +- **Context**: Queued tasks maintain the original request context +- **Database**: Uses the same database as other agent data + +## Limitations + +- Tasks are processed sequentially, not in parallel +- No priority system (FIFO only) +- Queue processing happens during agent execution, not as separate background jobs +- Failed tasks are dequeued after all retry attempts are exhausted (no dead-letter queue) diff --git a/docs/readonly-connections.md b/docs/readonly-connections.md new file mode 100644 index 0000000000..8ebc9ea2ed --- /dev/null +++ b/docs/readonly-connections.md @@ -0,0 +1,278 @@ +# Readonly Connections + +Readonly connections restrict certain WebSocket clients from modifying agent state while still letting them receive state updates and call non-mutating RPC methods. + +## Overview + +When a connection is marked as readonly: + +- It **receives** state updates from the server +- It **can call** RPC methods that don't modify state +- It **cannot** call `this.setState()` — neither via client-side `setState()` nor via a `@callable()` method that calls `this.setState()` internally + +```typescript +import { Agent, type Connection, type ConnectionContext } from "agents"; + +export class DocAgent extends Agent { + shouldConnectionBeReadonly(connection: Connection, ctx: ConnectionContext) { + const url = new URL(ctx.request.url); + return url.searchParams.get("mode") === "view"; + } +} +``` + +```typescript +// Client - view-only mode +const agent = useAgent({ + agent: "DocAgent", + name: "doc-123", + query: { mode: "view" }, + onStateUpdateError: (error) => { + toast.error("You're in view-only mode"); + } +}); +``` + +## Marking connections as readonly + +### On connect + +Override `shouldConnectionBeReadonly` to evaluate each connection when it first connects. Return `true` to mark it readonly. + +```typescript +export class MyAgent extends Agent { + shouldConnectionBeReadonly( + connection: Connection, + ctx: ConnectionContext + ): boolean { + const url = new URL(ctx.request.url); + const role = url.searchParams.get("role"); + return role === "viewer" || role === "guest"; + } +} +``` + +This hook runs before the initial state is sent to the client, so the connection is readonly from the very first message. + +### At any time + +Use `setConnectionReadonly` to change a connection's readonly status dynamically: + +```typescript +export class GameAgent extends Agent { + @callable() + async startSpectating() { + const { connection } = getCurrentAgent(); + if (connection) { + this.setConnectionReadonly(connection, true); + } + } + + @callable() + async joinAsPlayer() { + const { connection } = getCurrentAgent(); + if (connection) { + this.setConnectionReadonly(connection, false); + } + } +} +``` + +### Letting a connection toggle its own status + +A connection can toggle its own readonly status via a callable. This is useful for "lock/unlock" UIs where viewers can opt into editing mode: + +```typescript +import { Agent, callable, getCurrentAgent } from "agents"; + +export class CollabAgent extends Agent { + @callable() + async setMyReadonly(readonly: boolean) { + const { connection } = getCurrentAgent(); + if (connection) { + this.setConnectionReadonly(connection, readonly); + } + } +} +``` + +On the client: + +```typescript +// Toggle between readonly and writable +await agent.call("setMyReadonly", [true]); // lock +await agent.call("setMyReadonly", [false]); // unlock +``` + +### Checking status + +Use `isConnectionReadonly` to check a connection's current status: + +```typescript +@callable() +async getPermissions() { + const { connection } = getCurrentAgent(); + if (connection) { + return { canEdit: !this.isConnectionReadonly(connection) }; + } +} +``` + +## Handling errors on the client + +Errors surface in two ways depending on how the write was attempted: + +- **Client-side `setState()`** — the server sends a `cf_agent_state_error` message. Handle it with the `onStateUpdateError` callback. +- **`@callable()` methods** — the RPC call rejects with an error. Handle it with a `try`/`catch` around `agent.call()`. + +> **Note:** `onStateUpdateError` also fires when `validateStateChange` rejects a client-originated state update (with the message `"State update rejected"`). This makes the callback useful for handling any rejected state write, not just readonly errors. + +```typescript +const agent = useAgent({ + agent: "MyAgent", + name: "instance", + // Fires when client-side setState() is blocked + onStateUpdateError: (error) => { + setError(error); + } +}); + +// Fires when a callable that writes state is blocked +try { + await agent.call("updateSettings", [newSettings]); +} catch (e) { + setError(e instanceof Error ? e.message : String(e)); // "Connection is readonly" +} +``` + +To avoid showing errors in the first place, check permissions before rendering edit controls: + +```typescript +function Editor() { + const [canEdit, setCanEdit] = useState(false); + const agent = useAgent({ agent: "MyAgent", name: "instance" }); + + useEffect(() => { + agent.call("getPermissions").then((p) => setCanEdit(p.canEdit)); + }, []); + + return ; +} +``` + +## API reference + +### `shouldConnectionBeReadonly(connection, ctx)` + +Called when a connection is established. Override to control which connections are readonly. + +| Parameter | Type | Description | +| ------------ | ------------------- | ---------------------------- | +| `connection` | `Connection` | The connecting client | +| `ctx` | `ConnectionContext` | Contains the upgrade request | +| **Returns** | `boolean` | `true` to mark as readonly | + +Default: returns `false` (all connections are writable). + +### `setConnectionReadonly(connection, readonly?)` + +Mark or unmark a connection as readonly. Can be called at any time. + +| Parameter | Type | Description | +| ------------ | ------------ | ----------------------------------------- | +| `connection` | `Connection` | The connection to update | +| `readonly` | `boolean` | `true` to make readonly (default: `true`) | + +### `isConnectionReadonly(connection)` + +Check if a connection is currently readonly. + +| Parameter | Type | Description | +| ------------ | ------------ | ----------------------- | +| `connection` | `Connection` | The connection to check | +| **Returns** | `boolean` | `true` if readonly | + +### `onStateUpdateError` (client) + +Callback on `AgentClient` and `useAgent` options. Called when the server rejects a state update. + +| Parameter | Type | Description | +| --------- | -------- | ----------------------------- | +| `error` | `string` | Error message from the server | + +## How it works + +Readonly status is stored in the connection's WebSocket attachment, which persists through the WebSocket Hibernation API. The flag is namespaced internally so it cannot be accidentally overwritten by `connection.setState()`. This means: + +- **Survives hibernation** — the flag is serialized and restored when the agent wakes up +- **No cleanup needed** — connection state is automatically discarded when the connection closes +- **Zero overhead** — no database tables or queries, just the connection's built-in attachment +- **Safe from user code** — `connection.state` and `connection.setState()` never expose or overwrite the readonly flag + +When a readonly connection tries to modify state, the server blocks it — regardless of whether the write comes from client-side `setState()` or from a `@callable()` method: + +``` +Client (readonly) Agent + │ │ + │ setState({ count: 1 }) │ + │ ─────────────────────────────▶ │ Check readonly → blocked + │ ◀─────────────────────────── │ + │ cf_agent_state_error │ + │ │ + │ call("increment") │ + │ ─────────────────────────────▶ │ increment() calls this.setState() + │ │ Check readonly → throw + │ ◀─────────────────────────── │ + │ RPC error: "Connection is │ + │ readonly" │ + │ │ + │ call("getPermissions") │ + │ ─────────────────────────────▶ │ getPermissions() — no setState() + │ ◀─────────────────────────── │ + │ RPC result: { canEdit: false }│ +``` + +## What readonly does and does not restrict + +| Action | Allowed? | +| ------------------------------------------------------ | -------- | +| Receive state broadcasts | Yes | +| Call `@callable()` methods that don't write state | Yes | +| Call `@callable()` methods that call `this.setState()` | **No** | +| Send state updates via client-side `setState()` | **No** | + +The enforcement happens inside `setState()` itself. When a `@callable()` method tries to call `this.setState()` and the current connection context is readonly, the framework throws an `Error("Connection is readonly")`. This means you don't need manual permission checks in your RPC methods — any callable that writes state is automatically blocked for readonly connections. + +## Caveats + +### Side effects in callables still run + +The readonly check happens inside `this.setState()`, not at the start of the callable. If your method has side effects before the state write, those will still execute: + +```typescript +@callable() +async processOrder(orderId: string) { + await sendConfirmationEmail(orderId); // runs even for readonly connections + await chargePayment(orderId); // runs too + this.setState({ ...this.state, orders: [...this.state.orders, orderId] }); // throws +} +``` + +To avoid this, either check permissions before side effects or structure your code so the state write comes first: + +```typescript +@callable() +async processOrder(orderId: string) { + // Write state first — throws immediately for readonly connections + this.setState({ ...this.state, orders: [...this.state.orders, orderId] }); + // Side effects only run if setState succeeded + await sendConfirmationEmail(orderId); + await chargePayment(orderId); +} +``` + +## Related + +- [State Management](./state.md) +- [HTTP & WebSockets](./http-websockets.md) +- [Callable Methods](./callable-methods.md) diff --git a/docs/resumable-streaming.md b/docs/resumable-streaming.md new file mode 100644 index 0000000000..bb1452d811 --- /dev/null +++ b/docs/resumable-streaming.md @@ -0,0 +1,101 @@ +# Resumable Streaming + +The `AIChatAgent` class provides **automatic resumable streaming** out of the box. When a client disconnects and reconnects during an active stream, the response automatically resumes from where it left off. + +## How It Works + +When you use `AIChatAgent` with `useAgentChat`: + +1. **During streaming**: All chunks are automatically persisted to SQLite +2. **On disconnect**: The stream continues server-side, buffering chunks +3. **On reconnect**: Client requests a resume, receives all buffered chunks, and continues streaming + +No extra code is needed -- it just works. + +## Example + +### Server + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; +import { createWorkersAI } from "workers-ai-provider"; +import { streamText, convertToModelMessages } from "ai"; + +export class ChatAgent extends AIChatAgent { + async onChatMessage() { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + messages: await convertToModelMessages(this.messages) + }); + + // Automatic resumable streaming - no extra code needed + return result.toUIMessageStreamResponse(); + } +} +``` + +### Client + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ + agent: "ChatAgent", + name: "my-chat" + }); + + const { messages, sendMessage, status } = useAgentChat({ + agent + // resume: true is the default - streams automatically resume on reconnect + }); + + // ... render your chat UI +} +``` + +## Under the Hood + +### Server-side (`AIChatAgent`) + +- Creates SQLite tables for stream chunks and metadata on construction +- Each stream gets a unique ID and tracks chunk indices +- Chunks are batched (every 10 chunks) and flushed to SQLite for performance +- When a client sends `CF_AGENT_STREAM_RESUME_REQUEST`, the server checks for active streams and responds with `CF_AGENT_STREAM_RESUMING` +- Stale streams (older than 5 minutes) are cleaned up on restore +- Completed streams older than 24 hours are periodically garbage collected + +### Client-side (`useAgentChat`) + +- After the message handler is registered in `useEffect`, sends `CF_AGENT_STREAM_RESUME_REQUEST` to the server +- This avoids a race condition where the server's `onConnect` notification could arrive before the client's handler is ready +- On receiving `CF_AGENT_STREAM_RESUMING`, sends `CF_AGENT_STREAM_RESUME_ACK` +- Receives all buffered chunks with `replay: true` flag and applies them in a single batch +- Continues receiving live chunks as they arrive from the ongoing stream + +### The `replay` flag + +Replayed chunks include `replay: true` to distinguish them from live chunks. The client uses this to batch-apply all replayed chunks before rendering, which prevents intermediate states (like reasoning "Thinking..." indicators) from flashing briefly during replay. During a live stream, chunks arrive gradually and React renders each intermediate state naturally. + +## Disabling Resume + +If you do not want automatic resume (for example, for short responses), disable it: + +```tsx +const { messages } = useAgentChat({ + agent, + resume: false // Disable automatic stream resumption +}); +``` + +## Try It + +See [examples/resumable-stream-chat](../examples/resumable-stream-chat) for a complete working example. Start a long response, refresh the page mid-stream, and watch it resume automatically. + +## Related Docs + +- [Chat Agents](./chat-agents.md) — Full `AIChatAgent` and `useAgentChat` reference +- [Client Tools Continuation](./client-tools-continuation.md) — Client-side tool execution and auto-continuation diff --git a/docs/retries.md b/docs/retries.md new file mode 100644 index 0000000000..256c27f07d --- /dev/null +++ b/docs/retries.md @@ -0,0 +1,444 @@ +# Retries + +Retry failed operations with exponential backoff and jitter. The Agents SDK provides built-in retry support for scheduled tasks, queued tasks, and a general-purpose `this.retry()` method for your own code. + +## Overview + +Transient failures are common when calling external APIs, interacting with other services, or running background tasks. The retry system handles these automatically: + +- **Exponential backoff** — each retry waits longer than the last +- **Jitter** — randomized delays prevent thundering herd problems +- **Configurable** — tune attempts, delays, and caps per call site +- **Built-in** — schedule, queue, and workflow operations retry automatically + +## Quick Start + +Use `this.retry()` to retry any async operation: + +```typescript +import { Agent } from "agents"; + +export class MyAgent extends Agent { + async fetchWithRetry(url: string) { + const response = await this.retry(async () => { + const res = await fetch(url); + if (!res.ok) throw new Error(`HTTP ${res.status}`); + return res.json(); + }); + + return response; + } +} +``` + +By default, `this.retry()` makes up to 3 attempts with jittered exponential backoff. + +## `this.retry()` + +The `retry()` method is available on every `Agent` instance. It retries the provided function on any thrown error by default. + +```typescript +async retry( + fn: (attempt: number) => Promise, + options?: RetryOptions & { + shouldRetry?: (err: unknown, nextAttempt: number) => boolean; + } +): Promise +``` + +**Parameters:** + +- `fn` — the async function to retry. Receives the current attempt number (1-indexed). +- `options` — optional retry configuration (see [RetryOptions](#retryoptions) below). Options are validated eagerly — invalid values throw immediately. +- `options.shouldRetry` — optional predicate called with the thrown error and the next attempt number. Return `false` to stop retrying immediately. If not provided, all errors are retried. + +**Returns:** the result of `fn` on success. + +**Throws:** the last error if all attempts fail or `shouldRetry` returns `false`. + +### Examples + +**Basic retry:** + +```typescript +const data = await this.retry(() => fetch("https://api.example.com/data")); +``` + +**Custom retry options:** + +```typescript +const data = await this.retry( + async () => { + const res = await fetch("https://slow-api.example.com/data"); + if (!res.ok) throw new Error(`HTTP ${res.status}`); + return res.json(); + }, + { + maxAttempts: 5, + baseDelayMs: 500, + maxDelayMs: 10000 + } +); +``` + +**Using the attempt number:** + +```typescript +const result = await this.retry(async (attempt) => { + console.log(`Attempt ${attempt}...`); + return await this.callExternalService(); +}); +``` + +**Selective retry with `shouldRetry`:** + +Use `shouldRetry` to stop retrying on specific errors. The predicate receives both the error and the next attempt number: + +```typescript +const data = await this.retry( + async () => { + const res = await fetch("https://api.example.com/data"); + if (!res.ok) throw new HttpError(res.status, await res.text()); + return res.json(); + }, + { + maxAttempts: 5, + shouldRetry: (err, nextAttempt) => { + // Don't retry 4xx client errors — our request is wrong + if (err instanceof HttpError && err.status >= 400 && err.status < 500) { + return false; + } + return true; // retry everything else (5xx, network errors, etc.) + } + } +); +``` + +## Retries in Schedules + +Pass retry options when creating a schedule: + +```typescript +// Retry up to 5 times if the callback fails +await this.schedule( + 60, + "processTask", + { taskId: "123" }, + { + retry: { maxAttempts: 5 } + } +); + +// Retry with custom backoff +await this.schedule( + new Date("2026-03-01T09:00:00Z"), + "sendReport", + {}, + { + retry: { + maxAttempts: 3, + baseDelayMs: 1000, + maxDelayMs: 30000 + } + } +); + +// Cron with retries +await this.schedule( + "0 8 * * *", + "dailyDigest", + {}, + { + retry: { maxAttempts: 3 } + } +); + +// Interval with retries +await this.scheduleEvery( + 30, + "poll", + { source: "api" }, + { + retry: { maxAttempts: 5, baseDelayMs: 200 } + } +); +``` + +If the callback throws, it is retried according to the retry options. If all attempts fail, the error is logged and routed through `onError()`. The schedule is still removed (for one-time schedules) or rescheduled (for cron/interval) regardless of success or failure. + +## Retries in Queues + +Pass retry options when adding a task to the queue: + +```typescript +await this.queue( + "sendEmail", + { to: "user@example.com" }, + { + retry: { maxAttempts: 5 } + } +); + +await this.queue("processWebhook", webhookData, { + retry: { + maxAttempts: 3, + baseDelayMs: 500, + maxDelayMs: 5000 + } +}); +``` + +If the callback throws, it is retried before the task is dequeued. After all attempts are exhausted, the task is dequeued and the error is logged. + +## Validation + +Retry options are validated eagerly when you call `this.retry()`, `queue()`, `schedule()`, or `scheduleEvery()`. Invalid options throw immediately instead of failing later at execution time: + +```typescript +// Throws immediately: "retry.maxAttempts must be >= 1" +await this.queue("sendEmail", data, { + retry: { maxAttempts: 0 } +}); + +// Throws immediately: "retry.baseDelayMs must be > 0" +await this.schedule( + 60, + "process", + {}, + { + retry: { baseDelayMs: -100 } + } +); + +// Throws immediately: "retry.maxAttempts must be an integer" +await this.retry(() => fetch(url), { maxAttempts: 2.5 }); + +// Throws immediately: "retry.baseDelayMs must be <= retry.maxDelayMs" +// because baseDelayMs: 5000 exceeds the default maxDelayMs: 3000 +await this.queue("sendEmail", data, { + retry: { baseDelayMs: 5000 } +}); +``` + +Validation resolves partial options against class-level or built-in defaults before checking cross-field constraints. This means `{ baseDelayMs: 5000 }` is caught immediately when the resolved `maxDelayMs` is 3000, rather than failing later at execution time. + +## Default Behavior + +Even without explicit retry options, scheduled and queued callbacks are retried with sensible defaults: + +| Setting | Default | +| ------------- | ------- | +| `maxAttempts` | 3 | +| `baseDelayMs` | 100 | +| `maxDelayMs` | 3000 | + +These defaults apply to `this.retry()`, `queue()`, `schedule()`, and `scheduleEvery()`. Per-call-site options override them. + +### Class-Level Defaults + +Override the defaults for your entire agent via `static options`: + +```typescript +class MyAgent extends Agent { + static options = { + retry: { maxAttempts: 5, baseDelayMs: 200, maxDelayMs: 5000 } + }; +} +``` + +You only need to specify the fields you want to change — unset fields fall back to the built-in defaults: + +```typescript +class MyAgent extends Agent { + // Only override maxAttempts; baseDelayMs (100) and maxDelayMs (3000) stay default + static options = { + retry: { maxAttempts: 10 } + }; +} +``` + +Class-level defaults are used as fallbacks when a call site does not specify retry options. Per-call-site options always take priority: + +```typescript +// Uses class-level defaults (10 attempts) +await this.retry(() => fetch(url)); + +// Overrides to 2 attempts for this specific call +await this.retry(() => fetch(url), { maxAttempts: 2 }); +``` + +To disable retries for a specific task, set `maxAttempts: 1`: + +```typescript +await this.schedule( + 60, + "oneShot", + {}, + { + retry: { maxAttempts: 1 } + } +); +``` + +## RetryOptions + +```typescript +interface RetryOptions { + /** Maximum number of attempts (including the first). Must be an integer >= 1. Default: 3 */ + maxAttempts?: number; + /** Base delay in milliseconds for exponential backoff. Must be > 0 and <= maxDelayMs. Default: 100 */ + baseDelayMs?: number; + /** Maximum delay cap in milliseconds. Must be > 0. Default: 3000 */ + maxDelayMs?: number; +} +``` + +The delay between retries uses **full jitter exponential backoff**: + +``` +delay = random(0, min(2^attempt * baseDelayMs, maxDelayMs)) +``` + +This means early retries are fast (often under 200ms), and later retries back off to avoid overwhelming a failing service. The randomization (jitter) prevents multiple agents from retrying at the exact same moment. + +## How It Works + +### Backoff Strategy + +The retry system uses the "Full Jitter" strategy from the [AWS Architecture Blog](https://aws.amazon.com/blogs/architecture/exponential-backoff-and-jitter/). Given 3 attempts with default settings: + +| Attempt | Upper Bound | Actual Delay | +| ------- | ----------------------------- | ---------------- | +| 1 | min(2^1 \* 100, 3000) = 200ms | random(0, 200ms) | +| 2 | min(2^2 \* 100, 3000) = 400ms | random(0, 400ms) | +| 3 | (no retry — final attempt) | — | + +With `maxAttempts: 5` and `baseDelayMs: 500`: + +| Attempt | Upper Bound | Actual Delay | +| ------- | ----------------------------- | ----------------- | +| 1 | min(2 \* 500, 3000) = 1000ms | random(0, 1000ms) | +| 2 | min(4 \* 500, 3000) = 2000ms | random(0, 2000ms) | +| 3 | min(8 \* 500, 3000) = 3000ms | random(0, 3000ms) | +| 4 | min(16 \* 500, 3000) = 3000ms | random(0, 3000ms) | +| 5 | (no retry — final attempt) | — | + +### MCP Server Retries + +When adding an MCP server, you can configure retry options for connection and reconnection attempts: + +```typescript +await this.addMcpServer("github", "https://mcp.github.com", { + retry: { maxAttempts: 5, baseDelayMs: 1000, maxDelayMs: 10000 } +}); +``` + +These options are persisted and used when: + +- Restoring server connections after hibernation +- Establishing connections after OAuth completion + +Default: 3 attempts, 500ms base delay, 5s max delay. + +### Internal Retries + +The SDK also uses retries internally for platform operations: + +- **Workflow operations** (`terminateWorkflow`, `pauseWorkflow`, `resumeWorkflow`, `restartWorkflow`, `sendEventToWorkflow`) — retried with Durable Object-aware error detection. Transient DO errors are retried; overloaded errors are not. + +These internal retries use hardcoded defaults and are not configurable. + +## Patterns + +### Retry with Logging + +```typescript +class MyAgent extends Agent { + async resilientTask(payload: { url: string }) { + try { + const result = await this.retry( + async (attempt) => { + if (attempt > 1) { + console.log(`Retrying ${payload.url} (attempt ${attempt})...`); + } + const res = await fetch(payload.url); + if (!res.ok) throw new Error(`HTTP ${res.status}`); + return res.json(); + }, + { maxAttempts: 5 } + ); + console.log("Success:", result); + } catch (e) { + console.error("All retries failed:", e); + } + } +} +``` + +### Retry with Fallback + +```typescript +class MyAgent extends Agent { + async fetchData() { + try { + return await this.retry( + () => fetch("https://primary-api.example.com/data"), + { maxAttempts: 3, baseDelayMs: 200 } + ); + } catch { + // Primary failed, try fallback + return await this.retry( + () => fetch("https://fallback-api.example.com/data"), + { maxAttempts: 2 } + ); + } + } +} +``` + +### Combining Retries with Scheduling + +For operations that might take a long time to recover (minutes or hours), combine `this.retry()` for immediate retries with `this.schedule()` for delayed retries: + +```typescript +class MyAgent extends Agent { + async syncData(payload: { source: string; attempt?: number }) { + const attempt = payload.attempt ?? 1; + + try { + // Immediate retries for transient failures (seconds) + await this.retry(() => this.fetchAndProcess(payload.source), { + maxAttempts: 3, + baseDelayMs: 1000 + }); + } catch (e) { + if (attempt >= 5) { + console.error("Giving up after 5 scheduled attempts"); + return; + } + + // Schedule a retry in 5 minutes for longer outages + const delaySeconds = 300 * attempt; + await this.schedule(delaySeconds, "syncData", { + source: payload.source, + attempt: attempt + 1 + }); + console.log(`Scheduled retry ${attempt + 1} in ${delaySeconds}s`); + } + } +} +``` + +## Limitations + +- **No dead-letter queue.** If a queued or scheduled task fails all retry attempts, it is removed. Implement your own persistence if you need to track failed tasks. +- **Retry delays block the agent.** During the backoff delay, the Durable Object is awake but idle. For short delays (under 3 seconds) this is fine. For longer recovery times, use `this.schedule()` instead. +- **Queue retries are head-of-line blocking.** Queue items are processed sequentially. If one item is being retried with long delays, it blocks all subsequent items. If you need independent retry behavior, use `this.retry()` inside the callback rather than per-task retry options on `queue()`. +- **No circuit breaker.** The retry system does not track failure rates across calls. If a service is persistently down, each task will exhaust its retry budget independently. +- **`shouldRetry` is only available on `this.retry()`.** The `shouldRetry` predicate cannot be used with `schedule()` or `queue()` because functions cannot be serialized to the database. For scheduled/queued tasks, handle non-retryable errors inside the callback itself. + +## Related + +- [Scheduling](./scheduling.md) — schedule tasks for future execution +- [Queue](./queue.md) — background task queue +- [Workflows](./workflows.md) — durable multi-step processing with automatic retries diff --git a/docs/routing.md b/docs/routing.md new file mode 100644 index 0000000000..1f7cbdfb20 --- /dev/null +++ b/docs/routing.md @@ -0,0 +1,749 @@ +# Routing + +This guide explains how requests are routed to agents, how naming works, and patterns for organizing your agents. + +--- + +## How Routing Works + +When a request comes in, `routeAgentRequest()` examines the URL and routes it to the appropriate agent instance: + +``` +https://your-worker.dev/agents/{agent-name}/{instance-name} + └─────┬─────┘ └─────┬──────┘ + Class name Unique instance ID + (kebab-case) +``` + +**Example URLs:** + +| URL | Agent Class | Instance | +| -------------------------- | ----------- | ---------- | +| `/agents/counter/user-123` | `Counter` | `user-123` | +| `/agents/chat-room/lobby` | `ChatRoom` | `lobby` | +| `/agents/my-agent/default` | `MyAgent` | `default` | + +--- + +## Name Resolution + +Agent class names are automatically converted to kebab-case for URLs: + +| Class Name | URL Path | +| ------------- | -------------------------- | +| `Counter` | `/agents/counter/...` | +| `MyAgent` | `/agents/my-agent/...` | +| `ChatRoom` | `/agents/chat-room/...` | +| `AIAssistant` | `/agents/ai-assistant/...` | + +The router matches both the original name and kebab-case version, so these all work: + +- `useAgent({ agent: "Counter" })` → `/agents/counter/...` +- `useAgent({ agent: "counter" })` → `/agents/counter/...` + +--- + +## Basic Usage + +### `routeAgentRequest()` + +The main entry point for agent routing. Handles both HTTP requests and WebSocket upgrades: + +```typescript +import { routeAgentRequest } from "agents"; + +export default { + async fetch(request: Request, env: Env, ctx: ExecutionContext) { + // Route to agents - returns Response or undefined + const agentResponse = await routeAgentRequest(request, env); + + if (agentResponse) { + return agentResponse; + } + + // No agent matched - handle other routes + return new Response("Not found", { status: 404 }); + } +}; +``` + +### `getAgentByName()` + +Get a specific agent instance for server-side RPC calls or request forwarding: + +```typescript +import { getAgentByName, routeAgentRequest } from "agents"; + +export default { + async fetch(request: Request, env: Env) { + const url = new URL(request.url); + + // API endpoint that interacts with an agent + if (url.pathname === "/api/increment") { + const counter = await getAgentByName(env.Counter, "global-counter"); + const newCount = await counter.increment(); + return Response.json({ count: newCount }); + } + + // Regular agent routing + return ( + (await routeAgentRequest(request, env)) ?? + new Response("Not found", { status: 404 }) + ); + } +}; +``` + +--- + +## Instance Naming Patterns + +The instance name (the last part of the URL) determines which agent instance handles the request. Each unique name gets its own isolated agent with its own state. + +### Per-User Agents + +Each user gets their own agent instance: + +```typescript +// Client +const agent = useAgent({ + agent: "UserProfile", + name: `user-${userId}` // e.g., "user-abc123" +}); +``` + +``` +/agents/user-profile/user-abc123 → User abc123's agent +/agents/user-profile/user-xyz789 → User xyz789's agent (separate instance) +``` + +### Shared Rooms + +Multiple users share the same agent instance: + +```typescript +// Client +const agent = useAgent({ + agent: "ChatRoom", + name: roomId // e.g., "general" or "room-42" +}); +``` + +``` +/agents/chat-room/general → All users in "general" share this agent +``` + +### Global Singleton + +A single instance for the entire application: + +```typescript +// Client +const agent = useAgent({ + agent: "AppConfig", + name: "default" // Or any consistent name +}); +``` + +### Dynamic Naming + +Generate instance names based on context: + +```typescript +// Per-session +const agent = useAgent({ + agent: "Session", + name: sessionId +}); + +// Per-document +const agent = useAgent({ + agent: "Document", + name: `doc-${documentId}` +}); + +// Per-game +const agent = useAgent({ + agent: "Game", + name: `game-${gameId}-${Date.now()}` +}); +``` + +--- + +## Routing Options + +Both `routeAgentRequest()` and `getAgentByName()` accept options for customizing routing behavior. + +### CORS + +For cross-origin requests (common when your frontend is on a different domain): + +```typescript +const response = await routeAgentRequest(request, env, { + cors: true // Enable default CORS headers +}); +``` + +Or with custom CORS headers: + +```typescript +const response = await routeAgentRequest(request, env, { + cors: { + "Access-Control-Allow-Origin": "https://myapp.com", + "Access-Control-Allow-Methods": "GET, POST, OPTIONS", + "Access-Control-Allow-Headers": "Content-Type, Authorization" + } +}); +``` + +### Location Hints + +For latency-sensitive applications, hint where the agent should run: + +```typescript +// With getAgentByName +const agent = await getAgentByName(env.MyAgent, "instance-name", { + locationHint: "enam" // Eastern North America +}); + +// With routeAgentRequest (applies to all matched agents) +const response = await routeAgentRequest(request, env, { + locationHint: "enam" +}); +``` + +Available location hints: `wnam`, `enam`, `sam`, `weur`, `eeur`, `apac`, `oc`, `afr`, `me` + +### Jurisdiction + +For data residency requirements: + +```typescript +// With getAgentByName +const agent = await getAgentByName(env.MyAgent, "instance-name", { + jurisdiction: "eu" // EU jurisdiction +}); + +// With routeAgentRequest (applies to all matched agents) +const response = await routeAgentRequest(request, env, { + jurisdiction: "eu" +}); +``` + +### Props + +Since agents are instantiated by the runtime rather than constructed directly, `props` provides a way to pass initialization arguments: + +```typescript +const agent = await getAgentByName(env.MyAgent, "instance-name", { + props: { + userId: session.userId, + config: { maxRetries: 3 } + } +}); +``` + +Props are passed to the agent's `onStart` lifecycle method: + +```typescript +class MyAgent extends Agent { + private userId?: string; + private config?: { maxRetries: number }; + + async onStart(props?: { userId: string; config: { maxRetries: number } }) { + this.userId = props?.userId; + this.config = props?.config; + } +} +``` + +When using `props` with `routeAgentRequest`, the same props are passed to whichever agent matches the URL. This works well for universal context like authentication: + +```typescript +export default { + async fetch(request, env) { + const session = await getSession(request); + return routeAgentRequest(request, env, { + props: { userId: session.userId, role: session.role } + }); + } +}; +``` + +For agent-specific initialization, use `getAgentByName` instead where you control exactly which agent receives the props. + +> **Note:** For `McpAgent`, props are automatically stored and accessible via `this.props`. See [MCP Servers](/mcp-servers) for details. + +### Hooks + +`routeAgentRequest` supports hooks for intercepting requests before they reach agents: + +```typescript +const response = await routeAgentRequest(request, env, { + onBeforeConnect: (req, lobby) => { + // Called before WebSocket connections + // Return a Response to reject, Request to modify, or void to continue + }, + onBeforeRequest: (req, lobby) => { + // Called before HTTP requests + // Return a Response to reject, Request to modify, or void to continue + } +}); +``` + +These hooks are useful for authentication and validation. See [Securing Agents](/securing-agents) for detailed examples. + +--- + +## Custom URL Routing + +For advanced use cases where you need control over the URL structure, you can bypass the default `/agents/{agent}/{name}` pattern. + +### Using `basePath` (Client-Side) + +The `basePath` option lets clients connect to any URL path: + +```typescript +// Client connects to /user instead of /agents/user-agent/... +const agent = useAgent({ + agent: "UserAgent", // Required but ignored when basePath is set + basePath: "user" // → connects to /user +}); +``` + +This is useful when: + +- You want clean URLs without the `/agents/` prefix +- The instance name is determined server-side (e.g., from auth/session) +- You're integrating with an existing URL structure + +### Server-Side Instance Selection + +When using `basePath`, the server must handle routing. Use `getAgentByName()` to get the agent instance, then forward the request with `fetch()`: + +```typescript +export default { + async fetch(request: Request, env: Env) { + const url = new URL(request.url); + + // Custom routing - server determines instance from session + if (url.pathname === "/user") { + const session = await getSession(request); + const agent = await getAgentByName(env.UserAgent, session.userId); + return agent.fetch(request); // Forward request directly to agent + } + + // Default routing for standard /agents/... paths + return ( + (await routeAgentRequest(request, env)) ?? + new Response("Not found", { status: 404 }) + ); + } +}; +``` + +### Custom Path with Dynamic Instance + +Route different paths to different instances: + +```typescript +// Route /chat/{room} to ChatRoom agent +if (url.pathname.startsWith("/chat/")) { + const roomId = url.pathname.replace("/chat/", ""); + const agent = await getAgentByName(env.ChatRoom, roomId); + return agent.fetch(request); +} + +// Route /doc/{id} to Document agent +if (url.pathname.startsWith("/doc/")) { + const docId = url.pathname.replace("/doc/", ""); + const agent = await getAgentByName(env.Document, docId); + return agent.fetch(request); +} +``` + +### Receiving the Instance Identity (Client-Side) + +When using `basePath`, the client doesn't know which instance it connected to until the server tells it. The agent automatically sends its identity on connection: + +```typescript +const agent = useAgent({ + agent: "UserAgent", + basePath: "user", + onIdentity: (name, agentType) => { + console.log(`Connected to ${agentType} instance: ${name}`); + // e.g., "Connected to user-agent instance: user-123" + } +}); + +// Reactive state - re-renders when identity is received +return ( +
+ {agent.identified ? `Connected to: ${agent.name}` : "Connecting..."} +
+); +``` + +For `AgentClient`: + +```typescript +const agent = new AgentClient({ + agent: "UserAgent", + basePath: "user", + host: "example.com", + onIdentity: (name, agentType) => { + // Update UI with actual instance name + setInstanceName(name); + } +}); + +// Wait for identity before proceeding +await agent.ready; +console.log(agent.name); // Now has the server-determined name +``` + +### Handling Identity Changes on Reconnect + +If the identity changes on reconnect (e.g., session expired and user logs in as someone else), you can handle it with `onIdentityChange`: + +```typescript +const agent = useAgent({ + agent: "UserAgent", + basePath: "user", + onIdentityChange: (oldName, newName, oldAgent, newAgent) => { + console.log(`Session changed: ${oldName} → ${newName}`); + // Refresh state, show notification, etc. + } +}); +``` + +If `onIdentityChange` is not provided and identity changes, a warning is logged to help catch unexpected session changes. + +### Sub-Paths with `path` Option + +Append additional path segments to the URL: + +```typescript +// With basePath: /user/settings +useAgent({ agent: "UserAgent", basePath: "user", path: "settings" }); + +// Standard routing: /agents/my-agent/room/settings +useAgent({ agent: "MyAgent", name: "room", path: "settings" }); +``` + +### Disabling Identity for Security + +If your instance names contain sensitive data (session IDs, internal user IDs), you can disable identity sending: + +```typescript +class SecureAgent extends Agent { + // Don't expose instance names to clients + static options = { sendIdentityOnConnect: false }; +} +``` + +When identity is disabled: + +- `agent.identified` stays `false` +- `agent.ready` never resolves (use state updates instead) +- `onIdentity` and `onIdentityChange` are never called + +### When to Use Custom Routing + +| Scenario | Approach | +| --------------------------------- | --------------------------------------- | +| Standard agent access | Default `/agents/{agent}/{name}` | +| Instance from auth/session | `basePath` + `getAgentByName` + `fetch` | +| Clean URLs (no `/agents/` prefix) | `basePath` + custom routing | +| Legacy URL structure | `basePath` + custom routing | +| Complex routing logic | Custom routing in Worker | + +### Request Flow with Custom Routing + +``` +┌─────────────────┐ +│ /user request │ +│ (basePath) │ +└────────┬────────┘ + │ + ▼ +┌─────────────────┐ +│ Worker fetch │ +│ getSession() │ +└────────┬────────┘ + │ + ▼ +┌─────────────────┐ +│ getAgentByName │ +│ (session.userId)│ +└────────┬────────┘ + │ agent.fetch(request) + ▼ +┌─────────────────┐ +│ Agent Instance │ +│ onConnect() or │ +│ onRequest() │ +└─────────────────┘ +``` + +--- + +## Sub-Paths and HTTP Methods + +Requests can include sub-paths after the instance name. These are passed to your agent's `onRequest()` handler: + +``` +/agents/api/v1/users → agent: "api", instance: "v1", path: "/users" +/agents/api/v1/users/123 → agent: "api", instance: "v1", path: "/users/123" +``` + +Handle sub-paths in your agent: + +```typescript +export class API extends Agent { + async onRequest(request: Request): Promise { + const url = new URL(request.url); + + // url.pathname contains the full path including /agents/api/v1/... + // Extract the sub-path after your agent's base path + const path = url.pathname.replace(/^\/agents\/api\/[^/]+/, ""); + + if (request.method === "GET" && path === "/users") { + return Response.json(await this.getUsers()); + } + + if (request.method === "POST" && path === "/users") { + const data = await request.json(); + return Response.json(await this.createUser(data)); + } + + return new Response("Not found", { status: 404 }); + } +} +``` + +--- + +## Multiple Agents + +You can have multiple agent classes in one project. Each gets its own namespace: + +```typescript +// server.ts +export { Counter } from "./agents/counter"; +export { ChatRoom } from "./agents/chat-room"; +export { UserProfile } from "./agents/user-profile"; + +export default { + async fetch(request: Request, env: Env) { + return ( + (await routeAgentRequest(request, env)) ?? + new Response("Not found", { status: 404 }) + ); + } +}; +``` + +```jsonc +// wrangler.jsonc +{ + "durable_objects": { + "bindings": [ + { "name": "Counter", "class_name": "Counter" }, + { "name": "ChatRoom", "class_name": "ChatRoom" }, + { "name": "UserProfile", "class_name": "UserProfile" } + ] + }, + "migrations": [ + { + "tag": "v1", + "new_sqlite_classes": ["Counter", "ChatRoom", "UserProfile"] + } + ] +} +``` + +Each agent is accessed via its own path: + +``` +/agents/counter/... +/agents/chat-room/... +/agents/user-profile/... +``` + +--- + +## Routing with Authentication + +Check authentication before routing to agents: + +```typescript +export default { + async fetch(request: Request, env: Env) { + const url = new URL(request.url); + + // Protect agent routes + if (url.pathname.startsWith("/agents/")) { + const user = await authenticate(request, env); + if (!user) { + return new Response("Unauthorized", { status: 401 }); + } + + // Optionally, enforce that users can only access their own agents + const instanceName = url.pathname.split("/")[3]; + if (instanceName !== `user-${user.id}`) { + return new Response("Forbidden", { status: 403 }); + } + } + + return ( + (await routeAgentRequest(request, env)) ?? + new Response("Not found", { status: 404 }) + ); + } +}; +``` + +--- + +## Request Flow + +Here's how a request flows through the system: + +``` +┌─────────────────┐ +│ HTTP Request │ +│ or WebSocket │ +└────────┬────────┘ + │ + ▼ +┌─────────────────┐ +│ routeAgentRequest() +│ Parse URL path │ +└────────┬────────┘ + │ + ▼ +┌─────────────────┐ +│ Find binding in │ +│ env by name │ +└────────┬────────┘ + │ + ▼ +┌─────────────────┐ +│ Get/create DO │ +│ by instance ID │ +└────────┬────────┘ + │ + ▼ +┌─────────────────────────────────────┐ +│ Agent Instance │ +├─────────────────────────────────────┤ +│ WebSocket? → onConnect(), onMessage() +│ HTTP? → onRequest() │ +└─────────────────────────────────────┘ +``` + +--- + +## Troubleshooting + +### "Agent namespace not found" + +The error message lists available agents. Check: + +1. Agent class is exported from your entry point +2. Class name in code matches `class_name` in `wrangler.jsonc` +3. URL uses correct kebab-case name + +### Request returns 404 + +1. Verify the URL pattern: `/agents/{agent-name}/{instance-name}` +2. Check that `routeAgentRequest()` is called before your 404 handler +3. Ensure the response from `routeAgentRequest()` is returned (not just called) + +### WebSocket won't connect + +1. Don't modify the response from `routeAgentRequest()` for WebSocket upgrades +2. Ensure CORS is enabled if connecting from a different origin +3. Check browser dev tools for the actual error + +### `basePath` not working + +1. Ensure your Worker handles the custom path and forwards to the agent +2. Use `getAgentByName()` + `agent.fetch(request)` to forward requests +3. The `agent` parameter is still required but ignored when `basePath` is set +4. Check that the server-side route matches the client's `basePath` + +--- + +## API Reference + +### `routeAgentRequest(request, env, options?)` + +Routes a request to the appropriate agent. + +| Parameter | Type | Description | +| ------------------------- | ------------------------- | --------------------------------------------------- | +| `request` | `Request` | The incoming request | +| `env` | `Env` | Environment with agent bindings | +| `options.cors` | `boolean \| HeadersInit` | Enable CORS headers | +| `options.props` | `Record` | Props passed to whichever agent handles the request | +| `options.locationHint` | `string` | Preferred location for agent instances | +| `options.jurisdiction` | `string` | Data jurisdiction for agent instances | +| `options.onBeforeConnect` | `Function` | Callback before WebSocket connections | +| `options.onBeforeRequest` | `Function` | Callback before HTTP requests | + +**Returns:** `Promise` - Response if matched, undefined if no agent route + +### `getAgentByName(namespace, name, options?)` + +Get an agent instance by name for server-side RPC or request forwarding. + +| Parameter | Type | Description | +| ---------------------- | --------------------------- | --------------------------------------- | +| `namespace` | `DurableObjectNamespace` | Agent binding from env | +| `name` | `string` | Instance name | +| `options.locationHint` | `string` | Preferred location | +| `options.jurisdiction` | `string` | Data jurisdiction | +| `options.props` | `Record` | Initialization properties for `onStart` | + +**Returns:** `Promise>` - Typed stub for calling agent methods or forwarding requests + +### `useAgent(options)` / `AgentClient` Options + +Client connection options: + +| Option | Type | Description | +| ------------------ | ------------------------------------------------ | ---------------------------------------------------- | +| `agent` | `string` | Agent class name (required) | +| `name` | `string` | Instance name (default: `"default"`) | +| `basePath` | `string` | Full URL path - bypasses agent/name URL construction | +| `path` | `string` | Additional path to append to the URL | +| `onIdentity` | `(name, agent) => void` | Called when server sends identity | +| `onIdentityChange` | `(oldName, newName, oldAgent, newAgent) => void` | Called when identity changes on reconnect | + +**Return value properties (React hook):** + +| Property | Type | Description | +| ------------ | --------------- | --------------------------------------------- | +| `name` | `string` | Current instance name (reactive) | +| `agent` | `string` | Current agent class name (reactive) | +| `identified` | `boolean` | Whether identity has been received (reactive) | +| `ready` | `Promise` | Resolves when identity is received | + +### `Agent.options` (Server) + +Static options for agent configuration: + +| Option | Type | Default | Description | +| ---------------------------- | --------- | ------- | ---------------------------------------------------- | +| `hibernate` | `boolean` | `true` | Whether the agent should hibernate when inactive | +| `sendIdentityOnConnect` | `boolean` | `true` | Whether to send identity to clients on connect | +| `hungScheduleTimeoutSeconds` | `number` | `30` | Timeout before a running schedule is considered hung | + +```typescript +class SecureAgent extends Agent { + static options = { sendIdentityOnConnect: false }; +} +``` diff --git a/docs/scheduling.md b/docs/scheduling.md new file mode 100644 index 0000000000..fe45f47877 --- /dev/null +++ b/docs/scheduling.md @@ -0,0 +1,874 @@ +# Scheduling + +Schedule tasks to run in the future — whether that's seconds from now, at a specific date/time, or on a recurring cron schedule. Scheduled tasks survive agent restarts and are persisted to SQLite. + +## Overview + +The scheduling system supports four modes: + +| Mode | Syntax | Use Case | +| ------------- | ----------------------------------- | ------------------------- | +| **Delayed** | `this.schedule(60, ...)` | Run in 60 seconds | +| **Scheduled** | `this.schedule(new Date(...), ...)` | Run at specific time | +| **Cron** | `this.schedule("0 8 * * *", ...)` | Run on recurring schedule | +| **Interval** | `this.scheduleEvery(30, ...)` | Run every 30 seconds | + +Under the hood, scheduling uses [Durable Object alarms](https://developers.cloudflare.com/durable-objects/api/alarms/) to wake the agent at the right time. Tasks are stored in a SQLite table and executed in order. + +## Quick Start + +```typescript +import { Agent } from "agents"; + +export class ReminderAgent extends Agent { + async onRequest(request: Request) { + const url = new URL(request.url); + + // Schedule in 30 seconds + await this.schedule(30, "sendReminder", { + message: "Check your email" + }); + + // Schedule at specific time + await this.schedule(new Date("2025-02-01T09:00:00Z"), "sendReminder", { + message: "Monthly report due" + }); + + // Schedule recurring (every day at 8am) + await this.schedule("0 8 * * *", "dailyDigest", { + userId: url.searchParams.get("userId") + }); + + return new Response("Scheduled!"); + } + + async sendReminder(payload: { message: string }) { + console.log(`Reminder: ${payload.message}`); + // Send notification, email, etc. + } + + async dailyDigest(payload: { userId: string }) { + console.log(`Sending daily digest to ${payload.userId}`); + // Generate and send digest + } +} +``` + +## Scheduling Modes + +### Delayed Execution + +Pass a number to schedule a task to run after a delay in **seconds**: + +```typescript +// Run in 10 seconds +await this.schedule(10, "processTask", { taskId: "123" }); + +// Run in 5 minutes (300 seconds) +await this.schedule(300, "sendFollowUp", { email: "user@example.com" }); + +// Run in 1 hour +await this.schedule(3600, "checkStatus", { orderId: "abc" }); +``` + +**Use cases:** + +- Debouncing rapid events +- Delayed notifications ("You left items in your cart") +- Retry with backoff +- Rate limiting + +### Scheduled Execution + +Pass a `Date` object to schedule a task at a specific time: + +```typescript +// Run tomorrow at noon +const tomorrow = new Date(); +tomorrow.setDate(tomorrow.getDate() + 1); +tomorrow.setHours(12, 0, 0, 0); +await this.schedule(tomorrow, "sendReminder", { message: "Meeting time!" }); + +// Run at a specific timestamp +await this.schedule(new Date("2025-06-15T14:30:00Z"), "triggerEvent", { + eventId: "conference-2025" +}); + +// Run in 2 hours using Date math +const twoHoursFromNow = new Date(Date.now() + 2 * 60 * 60 * 1000); +await this.schedule(twoHoursFromNow, "checkIn", {}); +``` + +**Use cases:** + +- Appointment reminders +- Deadline notifications +- Scheduled content publishing +- Time-based triggers + +### Recurring (Cron) + +Pass a cron expression string for recurring schedules: + +```typescript +// Every day at 8:00 AM +await this.schedule("0 8 * * *", "dailyReport", {}); + +// Every hour +await this.schedule("0 * * * *", "hourlyCheck", {}); + +// Every Monday at 9:00 AM +await this.schedule("0 9 * * 1", "weeklySync", {}); + +// Every 15 minutes +await this.schedule("*/15 * * * *", "pollForUpdates", {}); + +// First day of every month at midnight +await this.schedule("0 0 1 * *", "monthlyCleanup", {}); +``` + +**Cron syntax:** `minute hour day month weekday` + +| Field | Values | Special Characters | +| ------------ | -------------- | ------------------ | +| Minute | 0-59 | `*` `,` `-` `/` | +| Hour | 0-23 | `*` `,` `-` `/` | +| Day of Month | 1-31 | `*` `,` `-` `/` | +| Month | 1-12 | `*` `,` `-` `/` | +| Day of Week | 0-6 (0=Sunday) | `*` `,` `-` `/` | + +**Common patterns:** + +```typescript +"* * * * *"; // Every minute +"*/5 * * * *"; // Every 5 minutes +"0 * * * *"; // Every hour (on the hour) +"0 0 * * *"; // Every day at midnight +"0 8 * * 1-5"; // Weekdays at 8am +"0 0 * * 0"; // Every Sunday at midnight +"0 0 1 * *"; // First of every month +``` + +**Use cases:** + +- Daily/weekly reports +- Periodic cleanup jobs +- Polling external services +- Health checks +- Subscription renewals + +### Interval + +Use `scheduleEvery()` to run a task at fixed intervals (in seconds). Unlike cron, intervals support sub-minute precision and arbitrary durations: + +```typescript +// Poll every 30 seconds +await this.scheduleEvery(30, "poll", { source: "api" }); + +// Health check every 45 seconds +await this.scheduleEvery(45, "healthCheck", {}); + +// Sync every 90 seconds (1.5 minutes - can't be expressed in cron) +await this.scheduleEvery(90, "syncData", { destination: "warehouse" }); +``` + +**Idempotency:** + +`scheduleEvery()` is idempotent on the combination of callback name, interval, and payload — calling it multiple times with the same arguments does not create duplicate schedules. This makes it safe to call in `onStart()`, which runs on every Durable Object wake: + +```typescript +class MyAgent extends Agent { + async onStart() { + // Safe: only one schedule is created, no matter how many times the DO wakes + await this.scheduleEvery(30, "tick"); + } + + async tick() { + console.log("tick", new Date().toISOString()); + } +} +``` + +Calling `scheduleEvery()` with a different interval or payload creates a separate schedule, even for the same callback: + +```typescript +// First call creates one schedule +await this.scheduleEvery(30, "poll"); + +// Second call with a different interval creates a second schedule +await this.scheduleEvery(60, "poll"); +// Two "poll" schedules exist: one every 30s and one every 60s + +// Third call with the same arguments as the first is a no-op +await this.scheduleEvery(30, "poll"); +// Still two schedules +``` + +Different callbacks also get their own independent schedules: + +```typescript +// These create two separate schedules (different callbacks) +await this.scheduleEvery(30, "poll"); +await this.scheduleEvery(30, "healthCheck"); +``` + +**Key differences from cron:** + +| Feature | Cron | Interval | +| ------------------- | ------------------------------ | ---------------------- | +| Minimum granularity | 1 minute | 1 second | +| Arbitrary intervals | No (must fit cron pattern) | Yes | +| Fixed schedule | Yes (e.g., "every day at 8am") | No (relative to start) | +| Overlap prevention | No | Yes (built-in) | + +**Overlap prevention:** + +If a callback takes longer than the interval, the next execution is skipped (not queued). This prevents runaway resource usage: + +```typescript +class PollingAgent extends Agent { + async poll() { + // If this takes 45 seconds and interval is 30 seconds, + // the next poll is skipped (with a warning logged) + const data = await slowExternalApi(); + await this.processData(data); + } +} + +// Set up 30-second interval +await this.scheduleEvery(30, "poll", {}); +``` + +When a skip occurs, you'll see a warning in logs: + +``` +Skipping interval schedule abc123: previous execution still running +``` + +**Error resilience:** + +If the callback throws an error, the interval continues — only that execution fails: + +```typescript +async syncData() { + // Even if this throws, the interval keeps running + const response = await fetch("https://api.example.com/data"); + if (!response.ok) throw new Error("Sync failed"); + // ... +} +``` + +**Use cases:** + +- Sub-minute polling (every 10, 30, 45 seconds) +- Intervals that don't map to cron (every 90 seconds, every 7 minutes) +- Rate-limited API polling with precise control +- Real-time data synchronization + +## Keeping the Agent Alive + +Durable Objects are evicted after a period of inactivity (typically 70-140 seconds with no incoming requests, WebSocket messages, or alarms). During long-running operations — streaming LLM responses, waiting on external APIs, running multi-step computations — the agent can be evicted mid-flight. + +`keepAlive()` prevents this by creating a 30-second heartbeat schedule that keeps the agent active until you are done: + +```typescript +const dispose = await this.keepAlive(); +try { + // Long-running work that must not be interrupted + const result = await longRunningComputation(); + await sendResults(result); +} finally { + dispose(); +} +``` + +The returned disposer function cancels the heartbeat. Always call it when the work is done — otherwise the heartbeat continues indefinitely. + +### keepAliveWhile() + +For scoped work, use `keepAliveWhile()` — it runs an async function and automatically cleans up the heartbeat when it completes (or throws): + +```typescript +const result = await this.keepAliveWhile(async () => { + const data = await longRunningComputation(); + return data; +}); +``` + +This is the recommended approach since you cannot forget to dispose the heartbeat. + +### How it works + +`keepAlive()` uses an in-memory reference count and the Durable Object alarm system directly. Each call increments the count; the disposer decrements it. While the count is above zero, `_scheduleNextAlarm()` ensures an alarm fires every 30 seconds, which resets the inactivity timer. No schedule rows are created and no observability events are emitted — the heartbeat is invisible to `getSchedules()` and the `agents:schedule` diagnostics channel. + +The heartbeat does not conflict with your own schedules — the alarm system multiplexes all schedules and the keepAlive heartbeat through a single alarm slot. + +### Multiple concurrent callers + +Each `keepAlive()` call returns an independent disposer: + +```typescript +const dispose1 = await this.keepAlive(); +const dispose2 = await this.keepAlive(); + +// Both heartbeats are active (ref count = 2) +dispose1(); // Decrements ref count to 1 +// Agent is still alive via dispose2's ref + +dispose2(); // Ref count reaches 0 — agent can go idle +``` + +### AIChatAgent + +`AIChatAgent` automatically calls `keepAlive()` during streaming responses. You do not need to add it yourself when using `AIChatAgent` — every LLM stream is protected from idle eviction by default. + +### When to use keepAlive() + +| Scenario | Use keepAlive()? | +| ------------------------------------------- | -------------------------------------- | +| Streaming LLM responses via `AIChatAgent` | No — already built in | +| Long-running computation in a custom Agent | Yes | +| Waiting on a slow external API call | Yes | +| Multi-step tool execution | Yes | +| Short request-response handlers | No — not needed | +| Background work via scheduling or workflows | No — alarms already keep the DO active | + +## Managing Schedules + +### Get a Schedule + +Retrieve a scheduled task by its ID: + +```typescript +const schedule = await this.getSchedule(scheduleId); + +if (schedule) { + console.log( + `Task ${schedule.id} will run at ${new Date(schedule.time * 1000)}` + ); + console.log(`Callback: ${schedule.callback}`); + console.log(`Type: ${schedule.type}`); // "scheduled" | "delayed" | "cron" | "interval" +} else { + console.log("Schedule not found"); +} +``` + +### List Schedules + +Query scheduled tasks with optional filters: + +```typescript +// Get all scheduled tasks +const allSchedules = this.getSchedules(); + +// Get only cron jobs +const cronJobs = this.getSchedules({ type: "cron" }); + +// Get tasks in the next hour +const upcoming = this.getSchedules({ + timeRange: { + start: new Date(), + end: new Date(Date.now() + 60 * 60 * 1000) + } +}); + +// Get a specific task by ID +const specific = this.getSchedules({ id: "abc123" }); + +// Combine filters +const upcomingCronJobs = this.getSchedules({ + type: "cron", + timeRange: { + start: new Date(), + end: new Date(Date.now() + 24 * 60 * 60 * 1000) + } +}); +``` + +### Cancel a Schedule + +Remove a scheduled task before it executes: + +```typescript +const cancelled = await this.cancelSchedule(scheduleId); + +if (cancelled) { + console.log("Schedule cancelled successfully"); +} else { + console.log("Schedule not found (may have already executed)"); +} +``` + +**Example: Cancellable reminders** + +```typescript +class ReminderAgent extends Agent { + async setReminder(userId: string, message: string, delaySeconds: number) { + const schedule = await this.schedule(delaySeconds, "sendReminder", { + userId, + message + }); + + // Store the schedule ID so user can cancel later + this.sql` + INSERT INTO user_reminders (user_id, schedule_id, message) + VALUES (${userId}, ${schedule.id}, ${message}) + `; + + return schedule.id; + } + + async cancelReminder(scheduleId: string) { + const cancelled = await this.cancelSchedule(scheduleId); + + if (cancelled) { + this.sql`DELETE FROM user_reminders WHERE schedule_id = ${scheduleId}`; + } + + return cancelled; + } + + async sendReminder(payload: { userId: string; message: string }) { + // Send the reminder... + + // Clean up the record + this.sql`DELETE FROM user_reminders WHERE user_id = ${payload.userId}`; + } +} +``` + +## The Schedule Object + +When you create or retrieve a schedule, you get a `Schedule` object: + +```typescript +type Schedule = { + id: string; // Unique identifier + callback: string; // Method name to call + payload: T; // Data passed to the callback + retry?: RetryOptions; // Retry options (if configured) + time: number; // Unix timestamp (seconds) of next execution +} & ( + | { type: "scheduled" } // One-time at specific date + | { type: "delayed"; delayInSeconds: number } // One-time after delay + | { type: "cron"; cron: string } // Recurring (cron expression) + | { type: "interval"; intervalSeconds: number } // Recurring (fixed interval) +); +``` + +**Example:** + +```typescript +const schedule = await this.schedule( + 60, + "myTask", + { foo: "bar" }, + { retry: { maxAttempts: 5 } } +); + +console.log(schedule); +// { +// id: "abc123xyz", +// callback: "myTask", +// payload: { foo: "bar" }, +// retry: { maxAttempts: 5 }, +// time: 1706745600, +// type: "delayed", +// delayInSeconds: 60 +// } +``` + +## Patterns + +### Rescheduling from Callbacks + +For dynamic recurring schedules, schedule the next run from within the callback: + +```typescript +class PollingAgent extends Agent { + async startPolling(intervalSeconds: number) { + await this.schedule(intervalSeconds, "poll", { interval: intervalSeconds }); + } + + async poll(payload: { interval: number }) { + try { + const data = await fetch("https://api.example.com/updates"); + await this.processUpdates(await data.json()); + } catch (error) { + console.error("Polling failed:", error); + } + + // Schedule the next poll (regardless of success/failure) + await this.schedule(payload.interval, "poll", payload); + } + + async stopPolling() { + // Cancel all polling schedules + const schedules = this.getSchedules({ type: "delayed" }); + for (const schedule of schedules) { + if (schedule.callback === "poll") { + await this.cancelSchedule(schedule.id); + } + } + } +} +``` + +### Retry on Failure + +For immediate retries (within seconds), use the built-in retry option: + +```typescript +// Retry up to 5 times with exponential backoff +await this.schedule( + 60, + "processTask", + { taskId: "123" }, + { + retry: { maxAttempts: 5 } + } +); +``` + +For longer recovery windows (minutes or hours), combine `this.retry()` for immediate retries with scheduled retries for extended outages: + +```typescript +class RetryAgent extends Agent { + async attemptTask(payload: { + taskId: string; + attempt: number; + maxAttempts: number; + }) { + try { + // Immediate retries for transient failures + await this.retry(() => this.doWork(payload.taskId), { + maxAttempts: 3 + }); + console.log( + `Task ${payload.taskId} succeeded on attempt ${payload.attempt}` + ); + } catch (error) { + if (payload.attempt >= payload.maxAttempts) { + console.error( + `Task ${payload.taskId} failed after ${payload.maxAttempts} attempts` + ); + return; + } + + // Schedule a retry in the future for longer outages + const delaySeconds = Math.pow(2, payload.attempt) * 60; + + await this.schedule(delaySeconds, "attemptTask", { + ...payload, + attempt: payload.attempt + 1 + }); + + console.log(`Scheduled retry in ${delaySeconds}s`); + } + } + + async doWork(taskId: string) { + // Your actual work here + } +} +``` + +See [Retries](./retries.md) for full documentation on retry options and patterns. + +### Self-Destructing Agents + +You can safely call `this.destroy()` from within a scheduled callback: + +```typescript +class TemporaryAgent extends Agent { + async onStart() { + // Self-destruct in 24 hours + await this.schedule(24 * 60 * 60, "cleanup", {}); + } + + async cleanup() { + // Perform final cleanup + console.log("Agent lifetime expired, cleaning up..."); + + // This is safe to call from a scheduled callback + await this.destroy(); + } +} +``` + +### Timezone-Aware Scheduling + +JavaScript Dates are UTC by default. For timezone-aware scheduling: + +```typescript +class TimezoneAgent extends Agent { + async scheduleForTimezone( + hour: number, + minute: number, + timezone: string, + callback: keyof this + ) { + // Create a date in the target timezone + const now = new Date(); + const formatter = new Intl.DateTimeFormat("en-US", { + timeZone: timezone, + year: "numeric", + month: "2-digit", + day: "2-digit", + hour: "2-digit", + minute: "2-digit", + hour12: false + }); + + // Parse and construct target time + const targetDate = new Date( + now.toLocaleString("en-US", { timeZone: timezone }) + ); + targetDate.setHours(hour, minute, 0, 0); + + // If time already passed today, schedule for tomorrow + if (targetDate <= now) { + targetDate.setDate(targetDate.getDate() + 1); + } + + return this.schedule(targetDate, callback, { timezone }); + } +} +``` + +## AI-Assisted Scheduling + +The SDK includes utilities for parsing natural language scheduling requests with AI. + +### getSchedulePrompt() + +Returns a system prompt for parsing natural language into scheduling parameters: + +```typescript +import { getSchedulePrompt, scheduleSchema } from "agents"; +import { generateObject } from "ai"; +import { openai } from "@ai-sdk/openai"; + +class SmartScheduler extends Agent { + async parseScheduleRequest(userInput: string) { + const result = await generateObject({ + model: openai("gpt-4o"), + system: getSchedulePrompt({ date: new Date() }), + prompt: userInput, + schema: scheduleSchema + }); + + return result.object; + } + + async handleUserRequest(input: string) { + // Parse: "remind me to call mom tomorrow at 3pm" + const parsed = await this.parseScheduleRequest(input); + + // parsed = { + // description: "call mom", + // when: { + // type: "scheduled", + // date: "2025-01-30T15:00:00Z" + // } + // } + + if (parsed.when.type === "scheduled" && parsed.when.date) { + await this.schedule(new Date(parsed.when.date), "sendReminder", { + message: parsed.description + }); + } else if (parsed.when.type === "delayed" && parsed.when.delayInSeconds) { + await this.schedule(parsed.when.delayInSeconds, "sendReminder", { + message: parsed.description + }); + } else if (parsed.when.type === "cron" && parsed.when.cron) { + await this.schedule(parsed.when.cron, "sendReminder", { + message: parsed.description + }); + } + } + + async sendReminder(payload: { message: string }) { + console.log(`Reminder: ${payload.message}`); + } +} +``` + +### scheduleSchema + +A Zod schema for validating parsed scheduling data: + +```typescript +import { scheduleSchema } from "agents"; + +// The schema uses a discriminated union on `when.type`: +// { +// description: string, +// when: +// | { type: "scheduled", date: string } // ISO 8601 date string +// | { type: "delayed", delayInSeconds: number } +// | { type: "cron", cron: string } +// | { type: "no-schedule" } +// } +``` + +When using this schema with OpenAI models via the AI SDK, you must pass `providerOptions: { openai: { strictJsonSchema: false } }` to `generateObject`. This is because the schema uses a discriminated union which is not compatible with OpenAI's strict structured outputs mode. + +## Scheduling vs Queue vs Workflows + +| Feature | Queue | Scheduling | Workflows | +| ------------------ | ------------------ | ----------------- | ------------------- | +| **When** | Immediately (FIFO) | Future time | Future time | +| **Execution** | Sequential | At scheduled time | Multi-step | +| **Retries** | Automatic | Automatic | Automatic | +| **Persistence** | SQLite | SQLite | Workflow engine | +| **Recurring** | No | Yes (cron) | No (use scheduling) | +| **Complex logic** | No | No | Yes | +| **Human approval** | No | No | Yes | + +**Use Queue when:** + +- You need background processing without blocking the response +- Tasks should run ASAP but don't need to block +- Order matters (FIFO) + +**Use Scheduling when:** + +- Tasks need to run at a specific time +- You need recurring jobs (cron) +- Delayed execution (debouncing, retries) + +**Use Workflows when:** + +- Multi-step processes with dependencies +- Automatic retries with backoff +- Human-in-the-loop approvals +- Long-running tasks (minutes to hours) + +## API Reference + +### schedule() + +```typescript +async schedule( + when: Date | string | number, + callback: keyof this, + payload?: T, + options?: { retry?: RetryOptions; idempotent?: boolean } +): Promise> +``` + +Schedule a task for future execution. + +**Parameters:** + +- `when` - When to execute: `number` (seconds delay), `Date` (specific time), or `string` (cron expression) +- `callback` - Name of the method to call +- `payload` - Data to pass to the callback (must be JSON-serializable) +- `options.retry` - Optional retry configuration. See [Retries](./retries.md) for details. +- `options.idempotent` - Deduplicate by callback + payload. Defaults to `true` for cron schedules, `false` for delayed and Date-based schedules. + +**Returns:** A `Schedule` object with the task details + +**Idempotency:** + +Cron schedules are idempotent by default — calling `schedule("0 * * * *", "tick")` multiple times with the same callback, cron expression, and payload returns the existing schedule instead of creating a duplicate. Set `idempotent: false` to override this. + +For delayed and Date-based schedules, set `idempotent: true` to opt in to the same dedup behavior (matched on callback + payload). This is especially useful when calling `schedule()` in `onStart()` to avoid accumulating duplicate rows across Durable Object restarts: + +```typescript +class MyAgent extends Agent { + async onStart() { + // Without idempotent: true, this creates a new row on every DO restart + await this.schedule(3600, "hourlyCleanup", {}, { idempotent: true }); + } +} +``` + +### scheduleEvery() + +```typescript +async scheduleEvery( + intervalSeconds: number, + callback: keyof this, + payload?: T, + options?: { retry?: RetryOptions } +): Promise> +``` + +Schedule a task to run repeatedly at a fixed interval. + +**Parameters:** + +- `intervalSeconds` - Number of seconds between executions (must be > 0) +- `callback` - Name of the method to call +- `payload` - Data to pass to the callback (must be JSON-serializable) +- `options.retry` - Optional retry configuration. See [Retries](./retries.md) for details. + +**Returns:** A `Schedule` object with `type: "interval"` + +**Behavior:** + +- **Idempotent on (callback, interval, payload)** — calling with the same callback, interval, and payload returns the existing schedule instead of creating a duplicate. A different interval or payload creates a new, independent schedule. +- First execution occurs after `intervalSeconds` (not immediately) +- If callback is still running when next execution is due, it's skipped (overlap prevention) +- If callback throws an error, the interval continues +- Cancel with `cancelSchedule(id)` to stop the entire interval + +### getSchedule() + +```typescript +getSchedule(id: string): Schedule | undefined +``` + +Get a scheduled task by ID. This method is synchronous. + +### getSchedules() + +```typescript +getSchedules(criteria?: { + id?: string; + type?: "scheduled" | "delayed" | "cron" | "interval"; + timeRange?: { start?: Date; end?: Date }; +}): Schedule[] +``` + +Get scheduled tasks matching the criteria. This method is synchronous. + +### cancelSchedule() + +```typescript +async cancelSchedule(id: string): Promise +``` + +Cancel a scheduled task. Returns `true` if cancelled, `false` if not found. + +### keepAlive() + +```typescript +async keepAlive(): Promise<() => void> +``` + +Create a 30-second heartbeat schedule that prevents the Durable Object from being evicted due to inactivity. Returns a disposer function that cancels the heartbeat when called. The disposer is idempotent — calling it multiple times is safe. + +See [Keeping the Agent Alive](#keeping-the-agent-alive) for usage details. + +### keepAliveWhile() + +```typescript +async keepAliveWhile(fn: () => Promise): Promise +``` + +Run an async function while keeping the Durable Object alive. The heartbeat is automatically started before the function runs and stopped when it completes (whether it succeeds or throws). Returns the value returned by the function. + +This is the recommended way to use keepAlive — it guarantees cleanup. + +## Limits + +- **Maximum tasks:** Limited by SQLite storage (each task is a row). Practical limit is tens of thousands per agent. +- **Task size:** Each task (including payload) can be up to 2MB. +- **Minimum delay:** 0 seconds (runs on next alarm tick) +- **Cron precision:** Minute-level (not seconds) +- **Interval precision:** Second-level +- **Cron jobs:** After execution, automatically rescheduled for the next occurrence +- **Interval jobs:** After execution, rescheduled for `now + intervalSeconds`; skipped if still running diff --git a/docs/securing-mcp-servers.md b/docs/securing-mcp-servers.md new file mode 100644 index 0000000000..027ca34624 --- /dev/null +++ b/docs/securing-mcp-servers.md @@ -0,0 +1,359 @@ +# Securing MCP Servers + +Model Context Protocol servers, like every other web application, need to be secured so they can be used by trusted users without abuse. The MCP spec uses the OAuth 2.1 standard for authentication between MCP clients and servers. + +Cloudflare's `workers-oauth-provider` lets you secure your MCP Server (or any application) running on a Cloudflare Worker. The provider handles token management, client registration, and access token validation automatically. + +```typescript +import { OAuthProvider, OAuthError } from "@cloudflare/workers-oauth-provider"; +import { createMcpHandler } from "agents/mcp"; + +// A Worker that exposes an MCP server +const apiHandler = { + async fetch(request: Request, env: unknown, ctx: ExecutionContext) { + return createMcpHandler(server)(request, env, ctx); + } +}; + +// Wrap with OAuth protection +export default new OAuthProvider({ + authorizeEndpoint: "/authorize", + tokenEndpoint: "/oauth/token", + clientRegistrationEndpoint: "/oauth/register", + + apiRoute: "/mcp", // Protected MCP endpoint + apiHandler: apiHandler, // Your MCP server + + defaultHandler: AuthHandler // Handles consent flow +}); +``` + +However, most MCP servers aren't just servers, they can actually be OAuth clients too. Your MCP server might sit between Claude Desktop and a third-party API like GitHub or Google. To Claude, you're a server. To GitHub, you're a client. This allows your users to authenticate and use their GitHub credentials to access your MCP server. We call this a proxy server. + +There are a few security footguns to securely building a proxy server. The rest of this document aims to outline best practises to securing an MCP server. + +## `redirect_uri` validation + +The `workers-oauth-provider` package handles this automatically. It validates that the `redirect_uri` in the authorization request matches one of the registered redirect URIs for the client. This prevents attackers from redirecting authorization codes to their own endpoints. + +## Consent dialog + +When your MCP server acts as an OAuth proxy to third-party providers (like Google, GitHub, etc.), you must implement your own consent dialog before forwarding users to the upstream authorization server. This prevents the ["confused deputy"](https://en.wikipedia.org/wiki/Confused_deputy_problem) problem where attackers could exploit cached consent from the third-party provider to gain unauthorized access. Your consent dialog should clearly identify the requesting MCP client by name and display the specific scopes being requested. Implementing this consent flow requires thinking about a few security concerns. + +### CSRF Protection + +Without CSRF protection, an attacker can trick users into approving malicious OAuth clients. Use a random token stored in a secure cookie and validate it on form submission. + +```typescript +// GET /authorize - Generate CSRF token when showing consent form +app.get("/authorize", async (c) => { + const { token: csrfToken, setCookie } = generateCSRFProtection(); + + return renderConsentDialog(c.req.raw, { + client: await c.env.OAUTH_PROVIDER.lookupClient(clientId), + csrfToken, // Pass to form as hidden field + setCookie // Set the cookie + // ... other dialog data + }); +}); + +// POST /authorize - Validate CSRF token when user approves +app.post("/authorize", async (c) => { + const formData = await c.req.raw.formData(); + + // Validate CSRF token exists and matches cookie + const { clearCookie } = validateCSRFToken(formData, c.req.raw); + + // Then redirect to upstream provider and clear the CSRF with the clearCookie header +}); + +// Helper functions +function generateCSRFProtection(): CSRFProtectionResult { + const token = crypto.randomUUID(); + const setCookie = `__Host-CSRF_TOKEN=${token}; HttpOnly; Secure; Path=/; SameSite=Lax; Max-Age=600`; + return { token, setCookie }; +} + +function validateCSRFToken( + formData: FormData, + request: Request +): ValidateCSRFResult { + const tokenFromForm = formData.get("csrf_token"); + const cookieHeader = request.headers.get("Cookie") || ""; + const tokenFromCookie = cookieHeader + .split(";") + .find((c) => c.trim().startsWith("__Host-CSRF_TOKEN=")) + ?.split("=")[1]; + + if (!tokenFromForm || !tokenFromCookie || tokenFromForm !== tokenFromCookie) { + throw new OAuthError("invalid_request", "CSRF token mismatch", 400); + } + + // Clear cookie after use (one-time use per RFC 9700) + return { + clearCookie: `__Host-CSRF_TOKEN=; HttpOnly; Secure; Path=/; SameSite=Lax; Max-Age=0` + }; +} +``` + +Include the token as a hidden field in your consent form: + +```html + +``` + +### Input sanitization + +User-controlled content (client names, logos, URIs) in your consent dialog can execute malicious scripts if not sanitized. Client registration is dynamic, so you must treat all client metadata as untrusted input. + +**Required protections:** + +- **Client names/descriptions**: HTML-escape all text before rendering (escape `<`, `>`, `&`, `"`, `'`) +- **Logo URLs**: Validate URL scheme (allow only `http:` and `https:`), reject `javascript:`, `data:`, `file:` schemes +- **Client URIs**: Same as logo URLs - whitelist http/https only +- **Scopes**: Treat as text, HTML-escape before display + +```typescript +function sanitizeText(text: string): string { + return text + .replace(/&/g, "&") + .replace(//g, ">") + .replace(/"/g, """) + .replace(/'/g, "'"); +} + +function sanitizeUrl(url: string): string { + if (!url) return ""; + try { + const parsed = new URL(url); + if (!["http:", "https:"].includes(parsed.protocol)) { + return ""; // Reject dangerous schemes + } + return url; + } catch { + return ""; // Invalid URL + } +} + +// Always sanitize before rendering +const clientName = sanitizeText(client.clientName); +const logoUrl = sanitizeText(sanitizeUrl(client.logoUri)); +``` + +### Content Security Policy (CSP) + +CSP headers instruct browsers to block dangerous content and behaviors. They provide defence in depth from multiple attack vectors. + +```typescript +function buildSecurityHeaders(setCookie: string, nonce?: string): HeadersInit { + const cspDirectives = [ + "default-src 'none'", // Deny everything by default + "script-src 'self'" + (nonce ? ` 'nonce-${nonce}'` : ""), // Allow scripts from same origin (+ nonce if using inline JS) + "style-src 'self' 'unsafe-inline'", // Allow inline styles for rendering + "img-src 'self' https:", // Allow client logos from HTTPS URLs + "font-src 'self'", // Allow web fonts from same origin + "form-action 'self'", // Only allow form submissions to same origin + "frame-ancestors 'none'", // Prevent clickjacking - block ALL iframe embedding + "base-uri 'self'", // Prevent base tag injection attacks + "connect-src 'self'" // Restrict fetch/XHR to same origin + ].join("; "); + + return { + "Content-Security-Policy": cspDirectives, + "X-Frame-Options": "DENY", // Legacy clickjacking protection for older browsers + "X-Content-Type-Options": "nosniff", // Prevent MIME sniffing attacks + "Content-Type": "text/html; charset=utf-8", + "Set-Cookie": setCookie + }; +} +``` + +### Inline JavaScript + +If your consent dialog needs inline JavaScript, use data attributes and nonces to prevent XSS attacks. + +```typescript +// Generate a unique nonce per request +const nonce = crypto.randomUUID(); + +const htmlContent = ` + + + +

Authorization approved! Redirecting...

+ + + +`; + +return new Response(htmlContent, { + headers: buildSecurityHeaders(setCookie, nonce) // Pass nonce to include in CSP +}); +``` + +- **Data attributes** store user-controlled data (like URLs) separately from JavaScript code, ensuring they're always treated as strings, never as executable code +- **Nonces** combined with the correct CSP headers (shown above) allow your specific inline script to execute while blocking any injected scripts + +Note: Frameworks such as Vite will automatically handle nonce generation and insertion for you. See their docs on [Content Security Policy](https://vite.dev/guide/features.html#content-security-policy-csp) for more information. + +## Handling State + +Between the consent dialog and the callback there is a gap where the user could do something nasty. We need to make sure it is the same user that hits authorize and then reaches back to our callback. Use a random state token stored server-side in KV with a short expiration time. + +```typescript +// Use in POST /authorize - after CSRF validation, before redirecting to upstream provider +// Firstly create a state token in KV +async function createOAuthState( + oauthReqInfo: AuthRequest, + kv: KVNamespace +): Promise<{ stateToken: string }> { + const stateToken = crypto.randomUUID(); + await kv.put(`oauth:state:${stateToken}`, JSON.stringify(oauthReqInfo), { + expirationTtl: 600 // 10 minutes + }); + return { stateToken }; +} + +// Bind state to browser session +async function bindStateToSession( + stateToken: string +): Promise<{ setCookie: string }> { + const consentedStateCookieName = "__Host-CONSENTED_STATE"; + + // Hash the state token to create a derived parameter + const encoder = new TextEncoder(); + const data = encoder.encode(stateToken); + const hashBuffer = await crypto.subtle.digest("SHA-256", data); + const hashArray = Array.from(new Uint8Array(hashBuffer)); + const hashHex = hashArray + .map((b) => b.toString(16).padStart(2, "0")) + .join(""); + + const setCookie = `${consentedStateCookieName}=${hashHex}; HttpOnly; Secure; Path=/; SameSite=Lax; Max-Age=600`; + return { setCookie }; +} + +// In the GET /callback - validate state from query params against both the KV and the session cookie before exchanging code +async function validateOAuthState( + request: Request, + kv: KVNamespace +): Promise<{ oauthReqInfo: AuthRequest; clearCookie: string }> { + const consentedStateCookieName = "__Host-CONSENTED_STATE"; + const url = new URL(request.url); + const stateFromQuery = url.searchParams.get("state"); + + if (!stateFromQuery) { + throw new OAuthError("invalid_request", "Missing state parameter", 400); + } + + // Check 1: Validate state exists in KV + const storedDataJson = await kv.get(`oauth:state:${stateFromQuery}`); + if (!storedDataJson) { + throw new OAuthError("invalid_request", "Invalid or expired state", 400); + } + + // Check 2: Validate state matches session cookie + const cookieHeader = request.headers.get("Cookie") || ""; + const cookies = cookieHeader.split(";").map((c) => c.trim()); + const consentedStateCookie = cookies.find((c) => + c.startsWith(`${consentedStateCookieName}=`) + ); + const consentedStateHash = consentedStateCookie + ? consentedStateCookie.substring(consentedStateCookieName.length + 1) + : null; + + if (!consentedStateHash) { + throw new OAuthError( + "invalid_request", + "Missing session binding cookie - authorization flow must be restarted", + 400 + ); + } + + // Hash the state from query and compare with cookie + const encoder = new TextEncoder(); + const data = encoder.encode(stateFromQuery); + const hashBuffer = await crypto.subtle.digest("SHA-256", data); + const hashArray = Array.from(new Uint8Array(hashBuffer)); + const stateHash = hashArray + .map((b) => b.toString(16).padStart(2, "0")) + .join(""); + + if (stateHash !== consentedStateHash) { + throw new OAuthError( + "invalid_request", + "State token does not match session - possible CSRF attack detected", + 400 + ); + } + + // Both checks passed - clean up KV and return the clear cookie header + await kv.delete(`oauth:state:${stateFromQuery}`); + const clearCookie = `${consentedStateCookieName}=; HttpOnly; Secure; Path=/; SameSite=Lax; Max-Age=0`; + + return { + oauthReqInfo: JSON.parse(storedDataJson), + clearCookie + }; +} +``` + +## Approved client + +MCP proxy servers must maintain a registry of approved client IDs per user and check this registry before initiating the third-party authorization flow. Store approved clients in a secure, cryptographically signed cookie with HMAC-SHA256. + +```typescript +// Use in POST /authorize - after user approves, add client to approved list +export async function addApprovedClient( + request: Request, + clientId: string, + cookieSecret: string +): Promise { + const existingApprovedClients = + (await getApprovedClientsFromCookie(request, cookieSecret)) || []; + const updatedApprovedClients = Array.from( + new Set([...existingApprovedClients, clientId]) + ); + + const payload = JSON.stringify(updatedApprovedClients); + const signature = await signData(payload, cookieSecret); // HMAC-SHA256 + const cookieValue = `${signature}.${btoa(payload)}`; + + return `__Host-APPROVED_CLIENTS=${cookieValue}; HttpOnly; Secure; Path=/; SameSite=Lax; Max-Age=2592000`; +} +``` + +When reading the cookie in GET /authorize (before showing the consent dialog), verify the signature before trusting the data. If the signature doesn't match or the client isn't in the list, show the consent dialog. If the client is approved, skip the dialog and proceed directly to creating the OAuth state. + +## Cookies + +### Why `__Host-` prefix? + +Throughout this document you'll see cookies named with the `__Host-` prefix (like `__Host-CSRF_TOKEN` and `__Host-APPROVED_CLIENTS`). This is especially important for MCP servers running on `*.workers.dev` domains. + +The `__Host-` prefix is a security feature that prevents subdomain attacks. When you set a cookie with this prefix: + +- It **must** be set with the `Secure` flag (HTTPS only) +- It **must** have `Path=/` +- It **must not** have a `Domain` attribute + +This means the cookie is locked to the exact domain that set it. Without `__Host-`, an attacker controlling `evil.workers.dev` could set cookies for your `mcp-server.workers.dev` domain and potentially inject malicious CSRF tokens or approved client lists. The `__Host-` prefix prevents this by ensuring only your specific domain can set and read these cookies. + +### Multiple OAuth clients on the same host + +If you're running multiple OAuth flows on the same domain (e.g., GitHub OAuth and Google OAuth on the same worker), namespace your cookies to prevent collisions. + +Instead of `__Host-CSRF_TOKEN`, use `__Host-CSRF_TOKEN_GITHUB` and `__Host-CSRF_TOKEN_GOOGLE`. Same applies for approved clients: `__Host-APPROVED_CLIENTS_GITHUB` vs `__Host-APPROVED_CLIENTS_GOOGLE`. This ensures each OAuth flow maintains isolated state. + +# More info + +- [MCP Authorization](https://modelcontextprotocol.io/specification/2025-06-18/basic/authorization) +- [MCP Security Best Practices](https://modelcontextprotocol.io/specification/draft/basic/security_best_practices) +- [RFC 9700 - Protecting Redirect Based Flows](https://www.rfc-editor.org/rfc/rfc9700#name-protecting-redirect-based-f) +- [RFC 9700 - Best Practices](https://www.rfc-editor.org/rfc/rfc9700#name-best-practices) diff --git a/docs/server-driven-messages.md b/docs/server-driven-messages.md new file mode 100644 index 0000000000..8b6a44c411 --- /dev/null +++ b/docs/server-driven-messages.md @@ -0,0 +1,401 @@ +# Trigger patterns + +Send messages and trigger LLM responses from the server without a human action. Use this for scheduled follow-ups, queue processing, email-triggered responses, and autonomous agent workflows. + +## Overview + +In a typical chat flow, the user sends a message and the agent responds. But agents often need to act on their own — a scheduled reminder fires, a webhook arrives, a workflow completes, or the agent decides to continue after inspecting its own response. + +The key primitives: + +| Primitive | Role | +| ------------------- | ---------------------------------------------------------------------------------- | +| `saveMessages` | Inject a message and trigger the LLM — the server-side equivalent of `sendMessage` | +| `persistMessages` | Store messages without triggering a response — for injecting context silently | +| `onChatResponse` | React when any response completes, including ones you did not initiate | +| `isServerStreaming` | Client-side flag: `true` when a server-initiated stream is active | + +### `saveMessages` vs `persistMessages` + +`saveMessages` persists messages to SQLite **and** triggers `onChatMessage` for a new LLM response. It is awaitable — after it returns, the LLM has responded and the message is persisted. + +`persistMessages` stores messages and broadcasts them to connected clients, but does **not** trigger a model turn. Use it when you want to inject context (for example, a system message or background data) into the conversation without starting a response. + +### When to use `saveMessages` vs `onChatResponse` + +**Use `saveMessages` when you control the trigger** — schedule callbacks, webhooks, email handlers, or any method where you decide when to inject a message. + +**Use `onChatResponse` when you need to react to responses you did not trigger** — user-initiated messages, auto-continuations after tool approvals, or any turn that the framework ran on your behalf. + +## `waitUntilStable` + +Always call `waitUntilStable()` before reading `this.messages` or calling `saveMessages` from schedule callbacks, webhooks, email handlers, or other non-chat entry points. + +`waitUntilStable()` waits until the conversation is fully stable: + +- No active LLM stream in progress +- No pending client-tool interactions (tool results or approvals the user has not yet provided) +- No queued continuation turns + +It returns `true` when stable, or `false` if the timeout expires before a pending interaction resolves. If nothing is pending, it returns immediately. + +```typescript +const stable = await this.waitUntilStable({ timeout: 30_000 }); +if (!stable) { + // The conversation is blocked on a user interaction or an in-flight + // stream that did not complete within 30 seconds. + console.warn("Conversation not stable, skipping server-driven message"); + return; +} +// Safe to read this.messages and call saveMessages. +``` + +Without this guard, you risk reading stale messages or overlapping with an in-flight stream. + +## Triggering responses from the server + +### Cron schedule + +A daily digest agent that summarizes activity every morning. Cron schedules are idempotent by default, so calling `schedule()` in `onStart` is safe — it will not create duplicates across Durable Object restarts. + +```typescript +import { AIChatAgent } from "@cloudflare/ai-chat"; + +export class DigestAgent extends AIChatAgent { + async onChatMessage() { + // ... your LLM call + } + + async onStart() { + await this.schedule("0 9 * * *", "dailyDigest"); + } + + async dailyDigest() { + const stable = await this.waitUntilStable({ timeout: 30_000 }); + if (!stable) { + console.warn("Conversation not stable, skipping daily digest"); + return; + } + + await this.saveMessages((messages) => [ + ...messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [ + { + type: "text", + text: "Summarize what happened since your last digest." + } + ], + createdAt: new Date() + } + ]); + // At this point the LLM has responded and the message is persisted. + } +} +``` + +The function form of `saveMessages` — `saveMessages((messages) => [...])` — reads the latest persisted messages at execution time. This avoids stale baselines when multiple calls queue up (for example, rapid webhook arrivals). See [scheduling](./scheduling.md) for more on `schedule()` and cron syntax. + +### Processing a queue + +When you control the trigger, a simple loop is the clearest pattern: + +```typescript +async processQueue() { + for (const task of this.taskQueue) { + const stable = await this.waitUntilStable({ timeout: 30_000 }); + if (!stable) { + console.warn("Conversation not stable, stopping queue processing"); + break; + } + + await this.saveMessages((messages) => [ + ...messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: task }], + createdAt: new Date() + } + ]); + // LLM has responded. this.messages is updated. Next iteration. + } + this.taskQueue = []; +} +``` + +No special hooks needed — `saveMessages` returns after the full turn completes. + +### Email-triggered + +```typescript +async onEmail(email: AgentEmail) { + const stable = await this.waitUntilStable({ timeout: 30_000 }); + if (!stable) { + console.warn("Conversation not stable, cannot process email"); + return; + } + + const subject = email.headers.get("subject") ?? "(no subject)"; + const body = await new Response(email.raw).text(); + + await this.saveMessages((messages) => [ + ...messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [ + { + type: "text", + text: `Email from ${email.from}: ${subject}\n\n${body}` + } + ], + createdAt: new Date() + } + ]); +} +``` + +### Webhook-triggered + +```typescript +async onRequest(request: Request): Promise { + const url = new URL(request.url); + + if (url.pathname.endsWith("/webhook") && request.method === "POST") { + const stable = await this.waitUntilStable({ timeout: 30_000 }); + if (!stable) { + return new Response("Agent is busy", { status: 503 }); + } + + const payload = await request.json(); + try { + await this.saveMessages((messages) => [ + ...messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [ + { type: "text", text: `Webhook event: ${JSON.stringify(payload)}` } + ], + createdAt: new Date() + } + ]); + return new Response("ok"); + } catch (error) { + console.error("Failed to process webhook:", error); + return new Response("Internal error", { status: 500 }); + } + } + + return super.onRequest(request); +} +``` + +### Injecting context without triggering a response + +Use `persistMessages` to add messages that the LLM will see on its next turn, without starting a turn now: + +```typescript +async addBackgroundContext(data: string) { + const stable = await this.waitUntilStable({ timeout: 30_000 }); + if (!stable) return; + + await this.persistMessages([ + ...this.messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [ + { type: "text", text: `[Background context]: ${data}` } + ], + createdAt: new Date() + } + ]); + // Message is stored and broadcast to clients, but no LLM call happens. +} +``` + +## Reacting to responses you did not initiate + +`onChatResponse` fires after **every** completed turn — user-initiated messages, `saveMessages` calls, and auto-continuations. Use it when you need to observe or react to responses regardless of how they were triggered. + +### Broadcasting state + +```typescript +import { AIChatAgent, type ChatResponseResult } from "@cloudflare/ai-chat"; + +export class ChatAgent extends AIChatAgent { + async onChatMessage() { + // ... your LLM call + } + + protected async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed") { + this.broadcast(JSON.stringify({ streaming: false })); + } + } +} +``` + +### Analytics + +```typescript +protected async onChatResponse(result: ChatResponseResult) { + try { + await fetch("https://analytics.example.com/event", { + method: "POST", + body: JSON.stringify({ + requestId: result.requestId, + status: result.status, + continuation: result.continuation + }) + }); + } catch (error) { + console.error("Analytics reporting failed:", error); + } +} +``` + +### Chained reasoning + +An agent can inspect its own response and decide whether to continue. This works for user-initiated messages too — you cannot predict what the user will ask, but you can react to what the agent said. + +```typescript +protected async onChatResponse(result: ChatResponseResult) { + if (result.status !== "completed") return; + + const lastText = result.message.parts + .filter((p) => p.type === "text") + .map((p) => p.text) + .join(""); + + if (lastText.includes("[NEEDS_MORE_RESEARCH]")) { + await this.saveMessages((messages) => [ + ...messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Continue your research." }], + createdAt: new Date() + } + ]); + } +} +``` + +When `saveMessages` is called from inside `onChatResponse`, the inner turn runs to completion and `saveMessages` returns. After the current `onChatResponse` call returns, the framework fires `onChatResponse` again for the inner response. This continues until no more work is queued. The framework never nests `onChatResponse` calls — results are drained sequentially. + +### Reactive queue processing + +When queue items can be added by external events (user messages, webhooks) at any time, `onChatResponse` lets you drain the queue after every response regardless of who triggered it: + +```typescript +protected async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed" && this.taskQueue.length > 0) { + const next = this.taskQueue.shift()!; + await this.saveMessages((messages) => [ + ...messages, + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: next }], + createdAt: new Date() + } + ]); + } +} +``` + +### `ChatResponseResult` fields + +| Field | Type | Description | +| -------------- | ------------------------------------- | ---------------------------------------- | +| `message` | `UIMessage` | The finalized assistant message | +| `requestId` | `string` | Unique ID for this turn | +| `continuation` | `boolean` | `true` if this was an auto-continuation | +| `status` | `"completed" \| "error" \| "aborted"` | How the turn ended | +| `error` | `string \| undefined` | Error details when `status` is `"error"` | + +## Client-side: detecting server-initiated streams + +When the server triggers a stream via `saveMessages`, the AI SDK's `status` stays `"ready"` because the client did not initiate the request. The `useAgentChat` hook provides two additional flags to handle this: + +| Flag | What it tracks | +| ------------------- | --------------------------------------------------------------------------------------------------------- | +| `status` | AI SDK lifecycle: `"submitted"`, `"streaming"`, `"ready"`, `"error"` — only for client-initiated requests | +| `isServerStreaming` | `true` when a server-initiated stream is active | +| `isStreaming` | `true` when either client or server streaming is active — use this for a universal indicator | + +Use `isStreaming` for most UI concerns (disabling the send button, showing a loading indicator). Use `isServerStreaming` only when you need to distinguish between user-initiated and server-initiated streams (for example, to show a different indicator like "Agent is working in the background..."). + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "ChatAgent" }); + const { messages, sendMessage, isStreaming, isServerStreaming } = + useAgentChat({ agent }); + + return ( +
+ {messages.map((m) => ( +
{/* render message */}
+ ))} + + {isServerStreaming &&
Agent is working in the background...
} + {!isServerStreaming && isStreaming &&
Agent is responding...
} + +
{ + e.preventDefault(); + const input = e.currentTarget.elements.namedItem( + "input" + ) as HTMLInputElement; + sendMessage({ text: input.value }); + input.value = ""; + }} + > + + +
+
+ ); +} +``` + +When a server-driven response arrives while the user is idle, connected clients see the new messages appear in real time. The `isStreaming` flag transitions from `false` → `true` → `false` as the stream runs, so UI elements like the send button automatically disable and re-enable. + +## Interaction with `messageConcurrency` + +The `messageConcurrency` setting on `AIChatAgent` controls how overlapping user submissions behave (`"queue"`, `"latest"`, `"merge"`, `"drop"`, `"debounce"`). This setting only applies to `sendMessage()` — user-initiated messages from the client. + +`saveMessages()` always uses serialized (queued) behavior regardless of the `messageConcurrency` setting. This means server-driven messages never get dropped, merged, or debounced — they always queue up and execute in order. + +## Combining with other Agent primitives + +| Primitive | How to combine | +| ------------------ | --------------------------------------------------------------------------------------------- | +| `schedule()` | Schedule a callback that calls `saveMessages` — see the cron example above | +| `queue()` | Queue a method that calls `saveMessages` for deferred processing | +| `runWorkflow()` | Start a Workflow; use `AgentWorkflow.agent` RPC to call a method that triggers `saveMessages` | +| `onEmail()` | Convert email content to a chat message and call `saveMessages` | +| `onRequest()` | Handle webhooks and call `saveMessages` | +| `this.broadcast()` | Broadcast custom state from `onChatResponse` | + +## Important notes + +- **`saveMessages` is awaitable.** After it returns, the LLM has responded and the message is persisted. Use this when you control the trigger. +- **Use the function form of `saveMessages`.** `saveMessages((messages) => [...messages, newMsg])` reads the latest persisted messages at execution time, avoiding stale baselines when multiple calls queue up. +- **`persistMessages` does not trigger a response.** Use it to inject context or system messages silently. +- **`onChatResponse` is for reacting to turns you did not initiate.** Use it for user-initiated messages, auto-continuations, or any turn where you did not call `saveMessages` yourself. +- **`onChatResponse` does not nest.** When `saveMessages` is called from inside `onChatResponse`, the inner turn completes and `onChatResponse` fires again sequentially — not recursively. +- **Messages are persisted before `onChatResponse` fires.** If the Durable Object evicts during the hook, the conversation is safe in SQLite — only the hook callback is lost. +- **`waitUntilStable()` before injecting.** Always call this from schedule callbacks, webhooks, or other non-chat entry points to avoid overlapping with an in-flight stream or pending tool interaction. +- **The client sees `done: true` before `onChatResponse` runs.** The server-side hook does not delay the client. +- **`messageConcurrency` does not affect `saveMessages`.** Server-driven messages always queue and execute in order. diff --git a/docs/sessions.md b/docs/sessions.md new file mode 100644 index 0000000000..8c48ee1ad9 --- /dev/null +++ b/docs/sessions.md @@ -0,0 +1,794 @@ +# Sessions (Experimental) + +The Session API provides persistent conversation storage for agents, with tree-structured messages, context blocks, compaction, full-text search, and AI-controllable tools. It runs entirely on Durable Object SQLite — no external database needed. + +> **Experimental.** The Session API is under `agents/experimental/memory/session`. The API surface is stable but may evolve before graduating to the main package. + +## Quick Start + +```typescript +import { Agent } from "agents"; +import { Session } from "agents/experimental/memory/session"; + +class MyAgent extends Agent { + session = Session.create(this) + .withContext("soul", { + provider: { get: async () => "You are a helpful assistant." } + }) + .withContext("memory", { + description: "Learned facts about the user", + maxTokens: 1100 + }) + .withCachedPrompt(); + + async onMessage(message) { + await this.session.appendMessage(message); + const history = this.session.getHistory(); + const system = await this.session.freezeSystemPrompt(); + const tools = await this.session.tools(); + // Pass history, system prompt, and tools to your LLM + } +} +``` + +## Session + +`Session` manages a single conversation's messages, context blocks, and compaction state. + +### Creating a Session + +There are two ways to create a Session: + +**Builder API (recommended)** — uses `Session.create(agent)` with a chainable builder. Context providers without an explicit `provider` option are auto-wired to SQLite. + +```typescript +const session = Session.create(this) + .withContext("soul", { provider: { get: async () => "You are helpful." } }) + .withContext("memory", { description: "Learned facts", maxTokens: 1100 }) + .withCachedPrompt() + .onCompaction(myCompactFn) + .compactAfter(100_000); +``` + +**Direct constructor** — takes a `SessionProvider` and options directly. Used when you want full control over providers. + +```typescript +import { + AgentSessionProvider, + AgentContextProvider +} from "agents/experimental/memory/session"; + +const session = new Session(new AgentSessionProvider(this), { + context: [ + { + label: "memory", + description: "Notes", + maxTokens: 500, + provider: new AgentContextProvider(this, "memory") + }, + { label: "soul", provider: { get: async () => "You are helpful." } } + ] +}); +``` + +### Builder Methods + +All builder methods return `this` for chaining. Order doesn't matter — providers are resolved lazily on first use. + +| Method | Description | +| ------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `Session.create(agent)` | Static factory. `agent` is any object with a `sql` tagged template method (i.e. your Agent/DO). | +| `.forSession(sessionId)` | Namespace this session by ID. Required for multi-session isolation when not using SessionManager. Context provider keys and storage are scoped to this ID. | +| `.withContext(label, options?)` | Add a context block. See [Context Blocks](#context-blocks). | +| `.withCachedPrompt(provider?)` | Enable system prompt persistence. The prompt is frozen on first use and survives DO hibernation/eviction. Without an explicit provider, auto-wires to SQLite. | +| `.onCompaction(fn)` | Register a compaction function. See [Compaction](#compaction). | +| `.compactAfter(tokenThreshold)` | Auto-compact when estimated token count exceeds the threshold. Checked after each `appendMessage()`. Requires `.onCompaction()`. | + +### Messages + +Messages use the `SessionMessage` type — a minimal shape with `id`, `role`, `parts`, and optional `createdAt`. The Vercel AI SDK's `UIMessage` is structurally compatible and can be passed directly without conversion. The session stores messages in a tree structure via `parent_id`, enabling branching conversations. + +```typescript +// Append — auto-parents to the latest leaf unless parentId is specified +await session.appendMessage(message); +await session.appendMessage(message, parentId); + +// Update an existing message (matched by message.id) +session.updateMessage(message); + +// Delete specific messages +session.deleteMessages(["msg-1", "msg-2"]); + +// Clear all messages and skill state +session.clearMessages(); +``` + +> **Note:** `appendMessage()` is `async` because it may trigger auto-compaction. The underlying storage write is synchronous (SQLite), but the compaction step involves an LLM call. All other write methods (`updateMessage`, `deleteMessages`, `clearMessages`) are synchronous. + +#### Reading History + +```typescript +// Linear history from root to the latest leaf +const messages = session.getHistory(); + +// History to a specific leaf (for branching) +const branch = session.getHistory(leafId); + +// Get a single message +const msg = session.getMessage("msg-1"); + +// Get the newest message +const latest = session.getLatestLeaf(); + +// Count messages in path +const count = session.getPathLength(); +``` + +#### Branching + +Messages form a tree. When you `appendMessage` with a `parentId` that already has children, you create a branch. Use `getBranches()` to get all child messages branching from a given point: + +```typescript +// Get all child messages that branch from messageId (e.g. multiple responses to a user message) +const branches = session.getBranches(messageId); +``` + +This powers features like response regeneration — pass the user message ID to get both the original and regenerated responses. `getHistory(leafId)` walks the chosen path. + +### Search + +Full-text search over the conversation history using SQLite FTS5: + +```typescript +const results = session.search("deployment Friday", { limit: 10 }); +// Returns: Array<{ id, role, content, createdAt? }> +``` + +Uses porter stemming and unicode tokenization. The search covers all messages in the session. + +> **Note:** `search()` throws if the session provider doesn't support search. The built-in `AgentSessionProvider` supports it. + +### WebSocket Broadcasts + +When the Session's `agent` object has a `broadcast()` method (all `Agent` subclasses do), the Session automatically broadcasts status events over WebSocket after each write operation: + +- **`CF_AGENT_SESSION`** — phase (`"idle"` or `"compacting"`), `tokenEstimate`, `tokenThreshold` +- **`CF_AGENT_SESSION_ERROR`** — emitted on compaction failure + +This allows connected clients to display real-time token usage and compaction status. + +--- + +## Context Blocks + +Context blocks are persistent key-value sections injected into the system prompt. Each block has a **label**, optional **description**, and a **provider** that determines its behavior. + +### Provider Types + +There are four provider types, detected by duck-typing: + +| Provider | Interface | Behavior | AI Tool | +| --------------------------- | ------------------------------- | ------------------------------------------------------------------------------------------------ | ----------------------------------------------- | +| **ContextProvider** | `get()` | Read-only block in system prompt | — | +| **WritableContextProvider** | `get()` + `set()` | Writable via AI | `set_context` | +| **SkillProvider** | `get()` + `load()` + `set?()` | On-demand keyed documents. `get()` returns a metadata listing; `load(key)` fetches full content. | `load_context`, `unload_context`, `set_context` | +| **SearchProvider** | `get()` + `search()` + `set?()` | Full-text searchable entries. `get()` returns a summary; `search(query)` runs FTS5. | `search_context`, `set_context` | + +All providers also support an optional `init(label)` method, called before first use with the block's label. + +### Built-in Providers + +**`AgentContextProvider`** — SQLite-backed writable context. This is what you get by default when using the builder without an explicit provider. + +```typescript +import { AgentContextProvider } from "agents/experimental/memory/session"; + +// Explicit usage — key determines the SQLite row +new AgentContextProvider(this, "memory"); +``` + +**`R2SkillProvider`** — Cloudflare R2 bucket for on-demand document loading. Skills are listed in the system prompt as metadata; the model loads full content on demand via `load_context`. + +```typescript +import { R2SkillProvider } from "agents/experimental/memory/session"; + +Session.create(this).withContext("skills", { + provider: new R2SkillProvider(env.SKILLS_BUCKET, { prefix: "skills/" }) +}); +``` + +Descriptions are stored in R2 custom metadata (`description` key). + +**`AgentSearchProvider`** — SQLite FTS5 searchable context. Entries are indexed and searchable by the model via `search_context`. + +```typescript +import { AgentSearchProvider } from "agents/experimental/memory/session"; + +Session.create(this).withContext("knowledge", { + description: "Searchable knowledge base", + provider: new AgentSearchProvider(this) +}); +``` + +### Adding and Removing Context at Runtime + +Blocks can be added and removed dynamically after initialization — useful for extensions: + +```typescript +// Add a new block (auto-wires to SQLite if no provider given) +await session.addContext("extension-notes", { + description: "From extension X", + maxTokens: 500 +}); + +// Remove it +session.removeContext("extension-notes"); + +// Rebuild the system prompt to reflect changes +await session.refreshSystemPrompt(); +``` + +> **Note:** `addContext` and `removeContext` do NOT automatically update the frozen system prompt. You must call `refreshSystemPrompt()` afterward. + +### Reading Context Blocks + +```typescript +// Single block +const block = session.getContextBlock("memory"); +// block: { label, description?, content, tokens, maxTokens?, writable, isSkill, isSearchable } + +// All blocks +const blocks = session.getContextBlocks(); +``` + +### Writing to Context Blocks + +```typescript +// Replace content entirely +await session.replaceContextBlock("memory", "User likes coffee."); + +// Append content +await session.appendContextBlock("memory", "\nUser prefers dark roast."); +``` + +> **Note:** Writing to a context block updates the provider immediately but does NOT update the frozen system prompt snapshot. This is intentional — it preserves the LLM prefix cache. Call `refreshSystemPrompt()` when you want changes reflected in the prompt. + +### System Prompt + +The system prompt is built from all context blocks with headers and metadata: + +``` +══════════════════════════════════════════════ +SOUL (Identity) [readonly] +══════════════════════════════════════════════ +You are a helpful assistant. + +══════════════════════════════════════════════ +MEMORY (Learned facts) [45% — 495/1100 tokens] +══════════════════════════════════════════════ +User likes coffee. +User prefers dark roast. +``` + +```typescript +// Freeze — first call renders and persists, subsequent calls return the cached value +const prompt = await session.freezeSystemPrompt(); + +// Refresh — re-render from current block state and persist +const updated = await session.refreshSystemPrompt(); +``` + +The frozen prompt survives DO hibernation and eviction when `withCachedPrompt()` is enabled. After eviction, the next `freezeSystemPrompt()` call loads from SQLite rather than re-rendering. + +### Skills (Load/Unload) + +Skills are on-demand documents stored in a `SkillProvider` (e.g. R2). The model sees a metadata listing in the system prompt and can load full content on demand: + +```typescript +// Unload a skill to free context space (rewrites the tool result in history) +session.unloadSkill("skills", "api-reference"); + +// Check what's currently loaded +const loaded = session.getLoadedSkillKeys(); // Set<"skills:api-reference"> +``` + +After hibernation/eviction, loaded skills are reconstructed by scanning conversation history for `load_context` tool results. This means skill state survives restarts without additional storage. + +> **Weird:** The skill restoration scans the entire conversation history looking for `load_context` tool invocations in assistant messages with `state: "output-available"`. When you unload a skill, it doesn't delete the tool result — it rewrites the `output` field to `"[skill unloaded: key]"` in-place. This means the original loaded content is permanently lost from history after unload. + +--- + +## AI Tools + +Session automatically generates tools based on the provider types of your context blocks. Pass these to your LLM alongside your own tools. + +```typescript +const tools = await session.tools(); +// Merge with your own tools: +const allTools = { ...tools, ...myTools }; +``` + +### `set_context` + +Generated when any writable block exists. Writes to regular blocks, skill blocks (keyed), or search blocks (keyed). + +- For regular blocks: `{ label, content, action: "replace" | "append" }` +- For skill blocks: `{ label, key, content, description? }` +- For search blocks: `{ label, key, content }` + +Enforces `maxTokens` limits. Returns a usage string like `"Written to memory. Usage: 45% (495/1100 tokens)"`. + +### `load_context` + +Generated when any skill block exists. Loads full content by key from a `SkillProvider`. + +- Input: `{ label, key }` +- Returns the document content, or `"Not found: key"` + +### `unload_context` + +Generated alongside `load_context`. Frees context space by unloading a previously loaded skill. + +- Input: `{ label, key }` +- Rewrites the tool result in conversation history to a short marker +- The skill remains available for re-loading + +The tool's description dynamically lists currently loaded skills. + +### `search_context` + +Generated when any search block exists. Full-text search within a searchable context block. + +- Input: `{ label, query }` +- Returns top 10 results by FTS5 rank, or `"No results found."` + +### `session_search` + +Available on `SessionManager` only (not on individual sessions). Searches across all sessions. + +- Input: `{ query }` +- Returns results from all sessions, or `"No results found."` + +Use `{ ...sessionTools, ...manager.tools() }` to give the model both per-session and cross-session tools. + +--- + +## Compaction + +Compaction summarizes older messages to keep conversations within token limits. Original messages are preserved in SQLite — the summary is a non-destructive overlay applied at read time. + +### Setup + +```typescript +import { createCompactFunction } from "agents/experimental/memory/utils/compaction-helpers"; + +const session = Session.create(this) + .withContext("memory", { maxTokens: 1100 }) + .onCompaction( + createCompactFunction({ + summarize: (prompt) => + generateText({ model: myModel, prompt }).then((r) => r.text), + protectHead: 3, // Keep first 3 messages (default: 3) + tailTokenBudget: 20000, // Protect ~20K tokens at the tail (default: 20000) + minTailMessages: 2 // Always keep at least 2 tail messages (default: 2) + }) + ) + .compactAfter(100_000); // Auto-compact at 100K estimated tokens +``` + +### How It Works + +1. **Protect head** — first N messages are never compacted (default 3) +2. **Protect tail** — walk backward from the end, accumulating tokens up to a budget (default 20K tokens) +3. **Align boundaries** — shift boundaries to avoid splitting tool call/result pairs +4. **Summarize middle** — send the middle section to an LLM with a structured format (Topic, Key Points, Current State, Open Items) +5. **Store overlay** — saved in `assistant_compactions` table, keyed by `fromMessageId` and `toMessageId` +6. **Iterative** — on subsequent compactions, the existing summary is passed to the LLM to update rather than replace + +When `getHistory()` is called, compaction overlays are applied transparently — the compacted range is replaced by a synthetic message with id `compaction_`. + +### Manual Compaction + +```typescript +// Run registered compaction function +const result = await session.compact(); + +// Or manage overlays directly +session.addCompaction("Summary of messages 1-50", "msg-1", "msg-50"); +const overlays = session.getCompactions(); +``` + +### Auto-Compaction + +When `.compactAfter(threshold)` is set, `appendMessage()` checks the estimated token count after each write. If it exceeds the threshold, `compact()` is called automatically. Auto-compaction failure is non-fatal — the message is already saved. + +> **Note:** Token estimation is heuristic (not tiktoken). It uses `max(chars/4, words*1.3)` with 4 tokens per-message overhead. This is intentional — tiktoken would add 80-120MB heap overhead, which exceeds Cloudflare Workers' 128MB limit. + +> **Weird:** Compaction is iterative but single-overlay. Each new compaction extends from the earliest existing compaction's `fromMessageId` to the new end. So you always have at most one active compaction overlay per session, and it keeps growing. The previous compaction rows remain in the database but are superseded by the latest one (which covers a wider range). `getCompactions()` returns all of them, but `getHistory()` applies the latest one. + +--- + +## SessionManager + +`SessionManager` is a registry for multiple named sessions within a single Durable Object. It provides lifecycle management, convenience methods, and cross-session search. + +### Creating a SessionManager + +```typescript +import { SessionManager } from "agents/experimental/memory/session"; + +const manager = SessionManager.create(this) + .withContext("soul", { provider: { get: async () => "You are helpful." } }) + .withContext("memory", { description: "Learned facts", maxTokens: 1100 }) + .withCachedPrompt() + .onCompaction(myCompactFn) + .compactAfter(100_000) + .withSearchableHistory("history"); +``` + +Context blocks, prompt caching, and compaction settings are propagated to all sessions created through the manager. Provider keys are automatically namespaced by session ID (e.g. `memory_`). + +### Builder Methods + +| Method | Description | +| ------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | +| `SessionManager.create(agent)` | Static factory. | +| `.withContext(label, options?)` | Add context block template for all sessions. | +| `.withCachedPrompt(provider?)` | Enable prompt persistence for all sessions. | +| `.onCompaction(fn)` | Register compaction function for all sessions. | +| `.compactAfter(tokenThreshold)` | Auto-compact threshold for all sessions. | +| `.withSearchableHistory(label)` | Add a cross-session searchable history block to every session. The model can search past conversations from any session. | + +### Session Lifecycle + +```typescript +// Create a new session +const info = manager.create("My Chat"); +// info: { id, name, parent_session_id, model, source, input_tokens, output_tokens, estimated_cost, end_reason, created_at, updated_at } + +// Create with metadata +const info2 = manager.create("My Chat", { + parentSessionId: "parent-id", + model: "claude-sonnet-4-20250514", + source: "web" +}); + +// Get session metadata (null if not found) +const session = manager.get(sessionId); + +// List all sessions (ordered by updated_at DESC) +const sessions = manager.list(); + +// Rename +manager.rename(sessionId, "New Name"); + +// Delete (clears messages too) +manager.delete(sessionId); +``` + +### Accessing Sessions + +```typescript +// Get or create the Session instance for an ID +// Lazy — creates on first access, caches for subsequent calls +const session = manager.getSession(sessionId); +``` + +### Message Convenience Methods + +These delegate to the underlying Session but also update the session's `updated_at` timestamp: + +```typescript +// Append a single message +await manager.append(sessionId, message, parentId?); + +// Add or update (upsert) +await manager.upsert(sessionId, message, parentId?); + +// Batch append (auto-chains parent IDs) +await manager.appendAll(sessionId, messages, parentId?); + +// Read history +const history = manager.getHistory(sessionId, leafId?); + +// Message count +const count = manager.getMessageCount(sessionId); + +// Clear messages +manager.clearMessages(sessionId); + +// Delete specific messages +manager.deleteMessages(sessionId, ["msg-1"]); +``` + +### Forking + +Fork a session at a specific message — copies history up to that point into a new session: + +```typescript +const forked = await manager.fork(sessionId, atMessageId, "Forked Chat"); +// forked.parent_session_id === sessionId +``` + +> **Weird:** Fork copies messages with new UUIDs, not the original IDs. This means message IDs in the forked session won't match the original. The fork also doesn't copy compaction overlays — the forked session starts clean with the materialized history. + +### Compaction + +```typescript +// Add a compaction overlay +manager.addCompaction(sessionId, summary, fromId, toId); + +// Get overlays +const compactions = manager.getCompactions(sessionId); + +// Compact and split — marks old session as ended, creates a continuation +const continuation = await manager.compactAndSplit( + sessionId, + summary, + "Continued Chat" +); +// continuation.parent_session_id === sessionId +// Old session gets end_reason = "compaction" +``` + +`compactAndSplit` is different from regular compaction — it creates a new session with a summary message instead of an in-place overlay. The original session is marked with `end_reason: "compaction"`. + +### Usage Tracking + +```typescript +manager.addUsage(sessionId, inputTokens, outputTokens, cost); +// Increments input_tokens, output_tokens, and estimated_cost on the session row +``` + +### Cross-Session Search + +```typescript +// Search across all sessions (FTS5) +const results = manager.search("deployment Friday", { limit: 20 }); +// Returns: Array<{ id, role, content, createdAt }> + +// Get tools for the model (includes session_search) +const tools = manager.tools(); +``` + +> **Note:** `manager.search()` uses a separate FTS5 index (`assistant_fts`) from per-session search. Messages are indexed into this table by the `AgentSessionProvider` when appended. The `session_search` tool limits results to 10. + +> **Weird:** `manager.search()` silently returns an empty array on FTS5 query errors (malformed queries, etc.) rather than throwing. + +--- + +## Storage + +All storage is in Durable Object SQLite. Tables are created lazily on first use. + +### Tables + +**`assistant_messages`** — Tree-structured messages. + +| Column | Type | Notes | +| ------------ | -------- | ------------------------------------------------------ | +| `id` | TEXT PK | Message ID | +| `session_id` | TEXT | Empty string for single-session; set for multi-session | +| `parent_id` | TEXT | Parent message ID (null for roots) | +| `role` | TEXT | `user`, `assistant`, `system` | +| `content` | TEXT | JSON-serialized `SessionMessage` | +| `created_at` | DATETIME | Auto-set | + +**`assistant_compactions`** — Compaction overlays. + +| Column | Type | Notes | +| ----------------- | -------- | ------------------------ | +| `id` | TEXT PK | Random UUID | +| `session_id` | TEXT | Scoped to session | +| `summary` | TEXT | LLM-generated summary | +| `from_message_id` | TEXT | Start of compacted range | +| `to_message_id` | TEXT | End of compacted range | +| `created_at` | DATETIME | Auto-set | + +**`assistant_fts`** — FTS5 virtual table for message search. Tokenizer: `porter unicode61`. + +**`assistant_sessions`** — Session registry (SessionManager only). + +| Column | Type | Notes | +| ------------------- | -------- | -------------------------- | +| `id` | TEXT PK | Random UUID | +| `name` | TEXT | Display name | +| `parent_session_id` | TEXT | For forks/splits | +| `model` | TEXT | Optional model identifier | +| `source` | TEXT | Optional source identifier | +| `input_tokens` | INTEGER | Cumulative input tokens | +| `output_tokens` | INTEGER | Cumulative output tokens | +| `estimated_cost` | REAL | Cumulative cost | +| `end_reason` | TEXT | `"compaction"` when split | +| `created_at` | DATETIME | Auto-set | +| `updated_at` | DATETIME | Updated on message ops | + +**`cf_agents_context_blocks`** — Persistent context block storage (`AgentContextProvider`). + +**`cf_agents_search_entries`** + **`cf_agents_search_fts`** — Searchable context entries and FTS5 index (`AgentSearchProvider`). + +--- + +## Custom Providers + +You can implement any of the four provider interfaces to plug in your own storage: + +```typescript +// Read-only context +const myProvider: ContextProvider = { + get: async () => "Static content here" +}; + +// Writable context (enables set_context tool) +const myWritable: WritableContextProvider = { + get: async () => fetchFromMyDB(), + set: async (content) => saveToMyDB(content) +}; + +// Skill provider (enables load_context tool) +const mySkills: SkillProvider = { + get: async () => "- api-ref: API Reference\n- guide: User Guide", + load: async (key) => fetchDocument(key), + set: async (key, content, description) => + saveDocument(key, content, description) // optional +}; + +// Search provider (enables search_context tool) +const mySearch: SearchProvider = { + get: async () => "42 entries indexed", + search: async (query) => searchMyIndex(query), + set: async (key, content) => indexContent(key, content) // optional +}; +``` + +You can also implement `SessionProvider` to replace the SQLite storage entirely: + +```typescript +const myStorage: SessionProvider = { + getMessage(id) { ... }, + getHistory(leafId?) { ... }, + getLatestLeaf() { ... }, + getBranches(messageId) { ... }, + getPathLength(leafId?) { ... }, + appendMessage(message, parentId?) { ... }, + updateMessage(message) { ... }, + deleteMessages(messageIds) { ... }, + clearMessages() { ... }, + addCompaction(summary, fromId, toId) { ... }, + getCompactions() { ... }, + searchMessages(query, limit) { ... } // optional +}; +``` + +--- + +## Utilities + +Exported from `agents/experimental/memory/utils`: + +### Token Estimation + +```typescript +import { + estimateStringTokens, + estimateMessageTokens +} from "agents/experimental/memory/utils/tokens"; + +estimateStringTokens("Hello world"); // heuristic: max(chars/4, words*1.3) +estimateMessageTokens(messages); // sum with 4 tokens per-message overhead +``` + +### Compaction Helpers + +```typescript +import { + createCompactFunction, + isCompactionMessage, + sanitizeToolPairs, + alignBoundaryForward, + alignBoundaryBackward, + findTailCutByTokens, + computeSummaryBudget, + buildSummaryPrompt, + COMPACTION_PREFIX +} from "agents/experimental/memory/utils/compaction-helpers"; +``` + +- `createCompactFunction(options)` — Full compaction implementation. See [Compaction](#compaction). +- `isCompactionMessage(msg)` — Check if a message is a compaction overlay (id starts with `compaction_`). +- `sanitizeToolPairs(messages)` — Fix orphaned tool call/result pairs after compaction. Removes orphaned results and adds stub results for calls whose results were dropped. +- `alignBoundaryForward/Backward(messages, idx)` — Shift a boundary index to avoid splitting tool call/result groups. +- `findTailCutByTokens(messages, headEnd, budget, minMessages)` — Find where to stop compressing using a token budget. +- `computeSummaryBudget(messages)` — 20% of compressed content tokens (minimum 100). +- `buildSummaryPrompt(messages, previousSummary, budget)` — Structured prompt for LLM summarization. + +--- + +## Exports + +Everything is exported from `agents/experimental/memory/session`: + +```typescript +import { + // Core + Session, + SessionManager, + + // Providers + AgentSessionProvider, + AgentContextProvider, + AgentSearchProvider, + R2SkillProvider, + + // Type guards + isWritableProvider, + isSkillProvider, + isSearchProvider, + + // Types + type SessionMessage, + type SessionMessagePart, + type SessionContextOptions, + type SessionInfo, + type SessionManagerOptions, + type SessionOptions, + type ContextBlock, + type ContextConfig, + type ContextProvider, + type WritableContextProvider, + type SkillProvider, + type SearchProvider, + type SearchResult, + type SessionProvider, + type StoredCompaction, + type SqlProvider +} from "agents/experimental/memory/session"; +``` + +Compaction utilities from `agents/experimental/memory/utils/compaction-helpers`: + +```typescript +import { + createCompactFunction, + isCompactionMessage, + sanitizeToolPairs, + COMPACTION_PREFIX, + type CompactResult, + type CompactOptions +} from "agents/experimental/memory/utils/compaction-helpers"; +``` + +Token utilities from `agents/experimental/memory/utils/tokens`: + +```typescript +import { + estimateStringTokens, + estimateMessageTokens +} from "agents/experimental/memory/utils/tokens"; +``` + +--- + +## Gotchas and Quirks + +Things that might surprise you: + +1. **Lazy initialization.** Sessions created with the builder don't initialize until first use. The first call to any method (e.g. `getHistory()`) triggers `_ensureReady()`, which creates SQLite tables, resolves providers, loads context blocks, and restores skill state from history. This means the first operation is slower than subsequent ones. + +2. **Snapshot freezing is sticky.** `freezeSystemPrompt()` caches the result. Writing to a context block does NOT update the cached snapshot — you must explicitly call `refreshSystemPrompt()`. This is deliberate (LLM prefix cache optimization), but easy to miss. + +3. **`appendMessage` is async, other writes are sync.** `appendMessage` is async only because it may trigger auto-compaction (which calls an LLM). The actual SQLite write is synchronous. `updateMessage`, `deleteMessages`, and `clearMessages` are all synchronous. + +4. **Skills survive hibernation via history scanning.** On initialization, the session scans the entire conversation history looking for `load_context` tool results to reconstruct which skills are loaded. This is clever but means initialization cost scales with conversation length. + +5. **Compaction overlays are superseding, not stacking.** Each compaction extends from the earliest existing `fromMessageId`. So you always have one effective overlay that keeps growing. Old compaction rows remain in the database but are unused. `getCompactions()` returns all rows, which can be confusing. + +6. **Search is silently absent.** `session.search()` throws if the provider doesn't support search, but `manager.search()` swallows FTS5 errors and returns `[]`. The `searchMessages` method on `SessionProvider` is optional (`searchMessages?`). + +7. **Fork copies with new IDs.** When forking via `SessionManager.fork()`, all messages get new UUIDs. If you're storing message IDs externally (e.g. for bookmarks), they won't survive a fork. + +8. **`removeContext` doesn't fire skill unload callbacks.** If you remove a context block that had loaded skills, the skill tracking is cleaned up but the conversation history is NOT rewritten. The tool results from those skills remain in history with their full content. + +9. **FTS5 query sanitization.** Both `AgentSearchProvider.search()` and `SessionManager.search()` quote individual words to prevent FTS5 syntax injection. This means you can't use FTS5 operators like `OR`, `NOT`, or `NEAR` — they'll be treated as literal search terms. + +10. **Auto-compaction failure is silent.** When `compactAfter` triggers and the compaction function throws, the error is emitted via WebSocket broadcast but the `appendMessage` call still succeeds. The message is saved; only the compaction is skipped. diff --git a/docs/state.md b/docs/state.md new file mode 100644 index 0000000000..9bb2c67e22 --- /dev/null +++ b/docs/state.md @@ -0,0 +1,512 @@ +# State Management + +Agents provide built-in state management with automatic persistence and real-time synchronization across all connected clients. + +## Overview + +Agent state is: + +- **Persistent** - Automatically saved to SQLite, survives restarts and hibernation +- **Synchronized** - Changes broadcast to all connected WebSocket clients instantly +- **Bidirectional** - Both server and clients can update state +- **Type-safe** - Full TypeScript support with generics + +```typescript +import { Agent } from "agents"; + +type GameState = { + players: string[]; + score: number; + status: "waiting" | "playing" | "finished"; +}; + +export class GameAgent extends Agent { + // Default state for new agents + initialState: GameState = { + players: [], + score: 0, + status: "waiting" + }; + + // React to state changes + onStateChanged(state: GameState, source: Connection | "server") { + if (source !== "server" && state.players.length >= 2) { + // Client added a player, start the game + this.setState({ ...state, status: "playing" }); + } + } + + addPlayer(name: string) { + this.setState({ + ...this.state, + players: [...this.state.players, name] + }); + } +} +``` + +## Defining Initial State + +Use the `initialState` property to define default values for new agent instances: + +```typescript +type State = { + messages: Message[]; + settings: UserSettings; + lastActive: string | null; +}; + +export class ChatAgent extends Agent { + initialState: State = { + messages: [], + settings: { theme: "dark", notifications: true }, + lastActive: null + }; +} +``` + +### Type Safety + +The second generic parameter to `Agent` defines your state type: + +```typescript +// State is fully typed +export class MyAgent extends Agent { + initialState: MyState = { count: 0 }; + + increment() { + // TypeScript knows this.state is MyState + this.setState({ count: this.state.count + 1 }); + } +} +``` + +### When Initial State Applies + +Initial state is applied lazily on first access, not on every wake: + +1. **New agent** - `initialState` is used and persisted +2. **Existing agent** - Persisted state is loaded from SQLite +3. **No `initialState` defined** - `this.state` is `undefined` + +```typescript +async onStart() { + // Safe to access - returns initialState if new, or persisted state + console.log("Current count:", this.state.count); +} +``` + +## Reading State + +Access the current state via the `this.state` getter: + +```typescript +async onRequest(request: Request) { + // Read current state + const { players, status } = this.state; + + if (status === "waiting" && players.length < 2) { + return new Response("Waiting for players..."); + } + + return new Response(JSON.stringify(this.state)); +} +``` + +### Undefined State + +If you don't define `initialState`, `this.state` returns `undefined`: + +```typescript +export class MinimalAgent extends Agent { + // No initialState defined + + async onConnect(connection: Connection) { + if (!this.state) { + // First time - initialize state + this.setState({ initialized: true }); + } + } +} +``` + +## Updating State + +Use `setState()` to update state. This: + +1. Saves to SQLite (persistent) +2. Broadcasts to all connected clients +3. Triggers `onStateChanged()` (after broadcast; best-effort) + +```typescript +// Replace entire state +this.setState({ + players: ["Alice", "Bob"], + score: 0, + status: "playing" +}); + +// Update specific fields (spread existing state) +this.setState({ + ...this.state, + score: this.state.score + 10 +}); +``` + +### State Must Be Serializable + +State is stored as JSON, so it must be serializable: + +```typescript +// Good - plain objects, arrays, primitives +this.setState({ + items: ["a", "b", "c"], + count: 42, + active: true, + metadata: { key: "value" } +}); + +// Bad - functions, classes, circular references +this.setState({ + callback: () => {}, // Functions don't serialize + date: new Date(), // Becomes string, loses methods + self: this // Circular reference +}); + +// For dates, use ISO strings +this.setState({ + createdAt: new Date().toISOString() +}); +``` + +## Responding to State Changes + +Override `onStateChanged()` to react when state changes (notifications/side-effects): + +```typescript +onStateChanged(state: GameState, source: Connection | "server") { + console.log("State updated:", state); + console.log("Updated by:", source === "server" ? "server" : source.id); +} +``` + +## Validating State Updates + +If you want to validate or reject state updates, override `validateStateChange()`: + +- **Runs before persistence and broadcast** +- **Must be synchronous** +- **Throwing aborts the update** + +```typescript +validateStateChange(nextState: GameState, source: Connection | "server") { + // Example: reject negative scores + if (nextState.score < 0) { + throw new Error("score cannot be negative"); + } +} +``` + +> `onStateChanged()` is not intended for validation; it is a notification hook and should not block broadcasts. +> +> **Migration note:** `onStateChanged` replaces the deprecated `onStateUpdate` (server-side hook). If you're using `onStateUpdate` on your agent class, rename it to `onStateChanged` — the signature and behavior are identical. A console warning will fire once per class until you rename it. + +### The `source` Parameter + +The `source` tells you who triggered the update: + +| Value | Meaning | +| ------------ | ----------------------------------- | +| `"server"` | Agent called `setState()` | +| `Connection` | A client pushed state via WebSocket | + +This is useful for: + +- Avoiding infinite loops (don't react to your own updates) +- Validating client input +- Triggering side effects only on client actions + +```typescript +onStateChanged(state: State, source: Connection | "server") { + // Ignore server-initiated updates + if (source === "server") return; + + // A client updated state - validate and process + const connection = source; + console.log(`Client ${connection.id} updated state`); + + // Maybe trigger something based on the change + if (state.status === "submitted") { + this.processSubmission(state); + } +} +``` + +### Common Pattern: Client-Driven Actions + +```typescript +onStateChanged(state: State, source: Connection | "server") { + if (source === "server") return; + + // Client added a message + const lastMessage = state.messages[state.messages.length - 1]; + if (lastMessage && !lastMessage.processed) { + // Process and update + this.setState({ + ...state, + messages: state.messages.map(m => + m.id === lastMessage.id ? { ...m, processed: true } : m + ) + }); + } +} +``` + +## Client-Side State Sync + +State synchronizes automatically with connected clients. Both `useAgent` and `AgentClient` expose a `state` property that tracks the current agent state. See [Client SDK](./client-sdk.md) for full details. + +### React (useAgent) + +```tsx +import { useAgent } from "agents/react"; + +function GameUI() { + const agent = useAgent({ + agent: "game-agent", + name: "room-123" + }); + + // Read state directly — reactive, triggers re-render on change + // Push state to agent with spread for partial updates + const addPlayer = (name: string) => { + agent.setState({ + ...agent.state, + players: [...(agent.state?.players ?? []), name] + }); + }; + + return
Players: {agent.state?.players.join(", ")}
; +} +``` + +### Vanilla JS (AgentClient) + +```typescript +import { AgentClient } from "agents/client"; + +const client = new AgentClient({ + agent: "game-agent", + name: "room-123", + host: "your-worker.workers.dev" +}); + +await client.ready; + +// Read state directly +console.log("Score:", client.state?.score); + +// Push state update with spread for partial updates +client.setState({ ...client.state, score: 100 }); +``` + +### State Flow + +``` +┌─────────────────────────────────────────────────────────────┐ +│ Agent │ +│ ┌─────────────────────────────────────────────────────┐ │ +│ │ this.state │ │ +│ │ (persisted in SQLite) │ │ +│ └─────────────────────────────────────────────────────┘ │ +│ ▲ │ │ +│ │ setState() │ broadcast │ +│ │ ▼ │ +└───────────┼──────────────────────────────┼──────────────────┘ + │ │ + │ │ WebSocket + │ │ +┌───────────┴──────────────────────────────┴───────────────────┐ +│ Clients │ +│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │ +│ │ Client 1 │ │ Client 2 │ │ Client 3 │ │ +│ │ state │ │ state │ │ state │ │ +│ └──────────┘ └──────────┘ └──────────┘ │ +│ │ +│ Any client can call setState() to push updates │ +└──────────────────────────────────────────────────────────────┘ +``` + +## State from Workflows + +When using [Workflows](./workflows.md), you can update agent state from workflow steps: + +```typescript +// In your workflow +async run(event: AgentWorkflowEvent, step: AgentWorkflowStep) { + // Replace entire state + await step.updateAgentState({ status: "processing", progress: 0 }); + + // Merge partial updates (preserves other fields) + await step.mergeAgentState({ progress: 50 }); + + // Reset to initialState + await step.resetAgentState(); + + return result; +} +``` + +These are durable operations - they persist even if the workflow retries. + +## Patterns & Best Practices + +### Keep State Small + +State is broadcast to all clients on every change. For large data: + +```typescript +// Bad - storing large arrays in state +initialState = { + allMessages: [] // Could grow to thousands of items +}; + +// Good - store in SQL, keep state light +initialState = { + messageCount: 0, + lastMessageId: null +}; + +// Query SQL for full data +async getMessages(limit = 50) { + return this.sql`SELECT * FROM messages ORDER BY created_at DESC LIMIT ${limit}`; +} +``` + +### Optimistic Updates + +For responsive UIs, update client state immediately: + +```typescript +// Client-side +function sendMessage(text: string) { + const optimisticMessage = { + id: crypto.randomUUID(), + text, + pending: true + }; + + // Update immediately — agent.state updates optimistically + agent.setState({ + ...agent.state, + messages: [...(agent.state?.messages ?? []), optimisticMessage] + }); + + // Server will confirm/update +} + +// Server-side +onStateChanged(state: State, source: Connection | "server") { + if (source === "server") return; + + const pendingMessages = state.messages.filter(m => m.pending); + for (const msg of pendingMessages) { + // Validate and confirm + this.setState({ + ...state, + messages: state.messages.map(m => + m.id === msg.id ? { ...m, pending: false, timestamp: Date.now() } : m + ) + }); + } +} +``` + +### State vs SQL + +| Use State For | Use SQL For | +| ---------------------------------- | ----------------- | +| UI state (loading, selected items) | Historical data | +| Real-time counters | Large collections | +| Active session data | Relationships | +| Configuration | Queryable data | + +```typescript +export class ChatAgent extends Agent { + // State: current UI state + initialState = { + typing: [], + unreadCount: 0, + activeUsers: [] + }; + + // SQL: message history + async getMessages(limit = 100) { + return this.sql` + SELECT * FROM messages + ORDER BY created_at DESC + LIMIT ${limit} + `; + } + + async saveMessage(message: Message) { + this.sql` + INSERT INTO messages (id, text, user_id, created_at) + VALUES (${message.id}, ${message.text}, ${message.userId}, ${Date.now()}) + `; + // Update state for real-time UI + this.setState({ + ...this.state, + unreadCount: this.state.unreadCount + 1 + }); + } +} +``` + +### Avoid Infinite Loops + +Be careful not to trigger state updates in response to your own updates: + +```typescript +// Bad - infinite loop +onStateChanged(state: State) { + this.setState({ ...state, lastUpdated: Date.now() }); +} + +// Good - check source +onStateChanged(state: State, source: Connection | "server") { + if (source === "server") return; // Don't react to own updates + this.setState({ ...state, lastUpdated: Date.now() }); +} +``` + +## API Reference + +### Properties + +| Property | Type | Description | +| -------------- | ------- | ---------------------------- | +| `state` | `State` | Current state (getter) | +| `initialState` | `State` | Default state for new agents | + +### Methods + +| Method | Signature | Description | +| ---------------- | -------------------------------------------------------- | --------------------------------------------- | +| `setState` | `(state: State) => void` | Update state, persist, and broadcast | +| `onStateChanged` | `(state: State, source: Connection \| "server") => void` | Called after state is persisted and broadcast | + +### Workflow Step Methods + +| Method | Description | +| ------------------------------- | ------------------------------------- | +| `step.updateAgentState(state)` | Replace agent state from workflow | +| `step.mergeAgentState(partial)` | Merge partial state from workflow | +| `step.resetAgentState()` | Reset to `initialState` from workflow | + +## Next Steps + +- [Readonly Connections](./readonly-connections.md) - Restrict which connections can update state +- [Client SDK](./client-sdk.md) - Full client-side state sync documentation +- [Workflows](./workflows.md) - Durable state updates from workflows +- [SQL API](./sql.md) - When to use SQL instead of state diff --git a/docs/think/client-tools.md b/docs/think/client-tools.md new file mode 100644 index 0000000000..b2e00e6a55 --- /dev/null +++ b/docs/think/client-tools.md @@ -0,0 +1,198 @@ +# Client Tools + +Think supports tools that execute in the browser. The client sends tool schemas in the chat request body, Think merges them with server tools, and when the LLM calls a client tool, the call is routed to the client for execution. + +## How Client Tools Work + +1. The client sends tool schemas as part of the chat request body +2. Think merges client tools with server-side tools (workspace, `getTools()`, session, MCP) +3. The LLM calls a client tool — the tool call chunk is sent to the client over WebSocket +4. The client executes the tool and sends back a `CF_AGENT_TOOL_RESULT` message +5. Think persists the result, broadcasts `CF_AGENT_MESSAGE_UPDATED`, and optionally auto-continues + +Client tools are tools without an `execute` function on the server — they only have a schema. When the LLM produces a tool call for one of these, Think sends the call to the client instead of executing it server-side. + +## Defining Client Tools + +On the client, pass `clientTools` to `useAgentChat`: + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "MyAgent" }); + const { messages, sendMessage } = useAgentChat({ + agent, + clientTools: { + getUserTimezone: { + description: "Get the user's timezone from their browser", + parameters: {}, + execute: async () => { + return Intl.DateTimeFormat().resolvedOptions().timeZone; + } + }, + getClipboard: { + description: "Read text from the user's clipboard", + parameters: {}, + execute: async () => { + return navigator.clipboard.readText(); + } + } + } + }); + + // ... render chat UI +} +``` + +The `parameters` field is a JSON Schema object describing the tool's input. The `execute` function runs in the browser. + +## Tool Approval + +Tools can require user approval before execution. This works for both server-side and client-side tools. + +### Server-side approval + +Use `needsApproval` in the tool definition: + +```typescript +getTools(): ToolSet { + return { + calculate: tool({ + description: "Perform a calculation", + inputSchema: z.object({ + a: z.number(), + b: z.number(), + operator: z.enum(["+", "-", "*", "/"]) + }), + needsApproval: async ({ a, b }) => + Math.abs(a) > 1000 || Math.abs(b) > 1000, + execute: async ({ a, b, operator }) => { + const ops: Record number> = { + "+": (x, y) => x + y, "-": (x, y) => x - y, + "*": (x, y) => x * y, "/": (x, y) => x / y + }; + return { result: ops[operator](a, b) }; + } + }) + }; +} +``` + +When `needsApproval` returns `true`: + +1. Think sends the tool call to the client with a pending approval state +2. The conversation pauses +3. The client shows an approval UI and sends `CF_AGENT_TOOL_APPROVAL` (approve or deny) +4. If approved, the tool executes and the conversation continues +5. If denied, the denial reason is returned to the model as the tool result + +### Handling approvals on the client + +`useAgentChat` provides approval helpers: + +```tsx +const { messages, sendMessage, addToolResult } = useAgentChat({ + agent, + onToolCall: ({ toolCall }) => { + // Auto-approve safe tools + if (toolCall.toolName === "read") { + return { approve: true }; + } + // Others go through the UI approval flow + } +}); +``` + +See [Client Tools Continuation](../client-tools-continuation.md) for the full protocol reference. + +## Auto-Continuation + +After a client tool result is received, Think can automatically continue the conversation without a new user message. This is the default behavior — when all pending tool results are received, Think starts a new model turn with the tool results in context. + +The continuation turn has `continuation: true` in the `TurnContext`, which you can use in `beforeTurn` to adjust model or tool selection: + +```typescript +beforeTurn(ctx: TurnContext) { + if (ctx.continuation) { + return { model: this.cheapModel }; + } +} +``` + +## Message Concurrency + +The `messageConcurrency` property controls how overlapping user submits behave when a chat turn is already active. + +```typescript +import { Think } from "@cloudflare/think"; +import type { MessageConcurrency } from "@cloudflare/think"; + +export class MyAgent extends Think { + override messageConcurrency: MessageConcurrency = "queue"; // default + + getModel() { + /* ... */ + } +} +``` + +### Strategies + +| Strategy | Behavior | +| ----------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------ | +| `"queue"` | Queue every submit and process them in order. Default. | +| `"latest"` | Keep only the latest overlapping submit. Superseded submits still persist their user messages but do not start their own model turn. | +| `"merge"` | Like `"latest"`, but all overlapping user messages remain in the conversation history. The model sees them all in one turn. | +| `"drop"` | Ignore overlapping submits entirely. Messages are not persisted. | +| `{ strategy: "debounce", debounceMs?: number }` | Trailing-edge latest with a quiet window (default 750ms). | + +Concurrency strategies only apply to `submit-message` requests. Regenerations, tool continuations, approvals, clears, `saveMessages`, and `continueLastTurn` keep their serialized behavior. + +### Examples + +For a search-as-you-type UI where each keystroke sends a new query: + +```typescript +export class SearchAgent extends Think { + override messageConcurrency: MessageConcurrency = "latest"; + getModel() { + /* ... */ + } +} +``` + +For a collaborative editor where multiple users type simultaneously: + +```typescript +export class CollabAgent extends Think { + override messageConcurrency: MessageConcurrency = "merge"; + getModel() { + /* ... */ + } +} +``` + +For a debounced input where the model only responds after the user stops typing: + +```typescript +export class DebouncedAgent extends Think { + override messageConcurrency: MessageConcurrency = { + strategy: "debounce", + debounceMs: 1000 + }; + getModel() { + /* ... */ + } +} +``` + +## Multi-Tab Broadcast + +Think broadcasts streaming responses to all connected WebSocket clients. When multiple browser tabs are connected to the same agent: + +- All tabs see the streamed response in real time +- Tool call states (pending, result, approval) are broadcast to all tabs +- The tab that resumes a stream is excluded from the broadcast to avoid duplicates +- `CF_AGENT_MESSAGE_UPDATED` events are sent to all tabs after tool results and message persistence diff --git a/docs/think/getting-started.md b/docs/think/getting-started.md new file mode 100644 index 0000000000..5b33369ee3 --- /dev/null +++ b/docs/think/getting-started.md @@ -0,0 +1,313 @@ +# Getting Started with Think + +Build a chat agent with persistent memory, built-in file tools, and streaming — step by step. + +By the end of this tutorial you will have a Think agent that: + +- Streams responses to a React chat UI +- Has persistent memory the model can read and write +- Includes workspace file tools (read, write, edit, find, grep, delete) +- Supports custom server-side tools + +## Prerequisites + +- Node.js 24+ +- A Cloudflare account with Workers AI access +- Familiarity with TypeScript and Cloudflare Workers + +## 1. Create a project + +```sh +mkdir my-think-agent && cd my-think-agent +npm init -y +``` + +Install dependencies: + +```sh +npm install @cloudflare/think @cloudflare/ai-chat agents ai @cloudflare/shell zod workers-ai-provider react react-dom +npm install -D wrangler @cloudflare/vite-plugin @cloudflare/workers-types @vitejs/plugin-react @tailwindcss/vite tailwindcss typescript vite +``` + +## 2. Configure wrangler + +Create `wrangler.jsonc`: + +```jsonc +{ + "name": "my-think-agent", + "compatibility_date": "2026-01-28", + "compatibility_flags": ["nodejs_compat", "experimental"], + "ai": { "binding": "AI" }, + "assets": { + "not_found_handling": "single-page-application", + "run_worker_first": ["/agents/*"] + }, + "durable_objects": { + "bindings": [{ "class_name": "MyAgent", "name": "MyAgent" }] + }, + "migrations": [{ "new_sqlite_classes": ["MyAgent"], "tag": "v1" }], + "main": "src/server.ts" +} +``` + +The `"experimental"` compatibility flag is required for Think. + +Create `vite.config.ts`: + +```typescript +import { cloudflare } from "@cloudflare/vite-plugin"; +import tailwindcss from "@tailwindcss/vite"; +import react from "@vitejs/plugin-react"; +import { defineConfig } from "vite"; + +export default defineConfig({ + plugins: [react(), cloudflare(), tailwindcss()] +}); +``` + +Create `tsconfig.json`: + +```json +{ + "extends": "agents/tsconfig" +} +``` + +## 3. Define the agent + +Create `src/server.ts`: + +```typescript +import { Think } from "@cloudflare/think"; +import { createWorkersAI } from "workers-ai-provider"; +import { routeAgentRequest } from "agents"; + +export class MyAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + + getSystemPrompt() { + return "You are a helpful assistant with access to a workspace filesystem."; + } +} + +export default { + async fetch(request: Request, env: Env) { + return ( + (await routeAgentRequest(request, env)) || + new Response("Not found", { status: 404 }) + ); + } +} satisfies ExportedHandler; +``` + +This is a working agent. Think automatically provides: + +- WebSocket chat protocol (compatible with `useAgentChat`) +- Message persistence in SQLite +- Resumable streaming (page refresh replays buffered chunks) +- Workspace file tools (read, write, edit, list, find, grep, delete) +- Abort/cancel support +- Error handling with partial message persistence + +## 4. Connect a React client + +Create `src/client.tsx`: + +```tsx +import { createRoot } from "react-dom/client"; +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "MyAgent" }); + const { messages, sendMessage, status } = useAgentChat({ agent }); + + return ( +
+

Think Agent

+ +
+ {messages.map((msg) => ( +
+ {msg.role}:{" "} + {msg.parts.map((part, i) => + part.type === "text" ? {part.text} : null + )} +
+ ))} +
+ +
{ + e.preventDefault(); + const input = e.currentTarget.elements.namedItem( + "input" + ) as HTMLInputElement; + if (!input.value.trim()) return; + sendMessage({ text: input.value }); + input.value = ""; + }} + > + + +
+ +

Status: {status}

+
+ ); +} + +createRoot(document.getElementById("root")!).render(); +``` + +Create `index.html`: + +```html + + + + + + Think Agent + + +
+ + + +``` + +## 5. Run it + +```sh +npx vite dev +``` + +Open the browser and send a message. The agent responds with streaming text, and workspace file tools are available to the model automatically. + +## 6. Add persistent memory + +Override `configureSession` to give the model writable memory that survives restarts: + +```typescript +export class MyAgent extends Think { + getModel(): LanguageModel { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } + + configureSession(session: Session) { + return session + .withContext("soul", { + provider: { + get: async () => + "You are a helpful assistant. Remember important facts about the user." + } + }) + .withContext("memory", { + description: "Important facts about the user and conversation.", + maxTokens: 2000 + }) + .withCachedPrompt(); + } +} +``` + +Now the model sees a `MEMORY` section in its system prompt and gets a `set_context` tool to update it. Facts written to memory persist in SQLite and survive DO hibernation and restarts. + +When you use `configureSession`, the system prompt is built from context blocks rather than `getSystemPrompt()`. The `"soul"` block above acts as the system identity — it is read-only and always appears first. The `"memory"` block is writable, and the model proactively updates it when it learns something useful. + +See the [Sessions documentation](../sessions.md) for context blocks, compaction, search, skills, and multi-session support. + +## 7. Add custom tools + +Override `getTools()` to add your own tools alongside the built-in workspace tools: + +```typescript +import { tool } from "ai"; +import { z } from "zod"; + +export class MyAgent extends Think { + getModel(): LanguageModel { + /* ... */ + } + configureSession(session: Session) { + /* ... */ + } + + getTools(): ToolSet { + return { + getWeather: tool({ + description: "Get the current weather for a city", + inputSchema: z.object({ + city: z.string().describe("City name") + }), + execute: async ({ city }) => { + const res = await fetch( + `https://api.weatherapi.com/v1/current.json?key=${this.env.WEATHER_KEY}&q=${city}` + ); + return res.json(); + } + }) + }; + } +} +``` + +Think merges tools from multiple sources automatically. On every turn, the model has access to: + +1. **Workspace tools** — read, write, edit, list, find, grep, delete (built-in) +2. **Session tools** — set_context, load_context, search_context (from `configureSession`) +3. **Your tools** — from `getTools()` +4. **MCP tools** — from connected MCP servers (if any) +5. **Client tools** — from the browser (if any) + +## 8. Add lifecycle hooks + +Think provides hooks that fire on every turn, regardless of entry path: + +```typescript +import type { + TurnContext, + TurnConfig, + ChatResponseResult +} from "@cloudflare/think"; + +export class MyAgent extends Think { + getModel(): LanguageModel { + /* ... */ + } + + beforeTurn(ctx: TurnContext): TurnConfig | void { + console.log( + `Turn starting: ${Object.keys(ctx.tools).length} tools available` + ); + } + + onChatResponse(result: ChatResponseResult) { + console.log(`Turn ${result.status}: ${result.message.parts.length} parts`); + } +} +``` + +See [Lifecycle Hooks](./lifecycle-hooks.md) for the full reference. + +## Next Steps + +- [Lifecycle Hooks](./lifecycle-hooks.md) — control model behavior, switch models per-turn, restrict tools +- [Tools](./tools.md) — workspace tools, code execution, extensions +- [Client Tools](./client-tools.md) — browser-side tools, approval flows, concurrency +- [Sub-agents and Programmatic Turns](./sub-agents.md) — RPC streaming, scheduled turns, recovery +- [Sessions](../sessions.md) — context blocks, compaction, search, multi-session diff --git a/docs/think/index.md b/docs/think/index.md new file mode 100644 index 0000000000..0d24cedd5b --- /dev/null +++ b/docs/think/index.md @@ -0,0 +1,241 @@ +# Think (Experimental) + +`@cloudflare/think` is an opinionated chat agent base class for Cloudflare Workers. It handles the full chat lifecycle — agentic loop, message persistence, streaming, tool execution, client tools, stream resumption, and extensions — all backed by Durable Object SQLite. + +Think works as both a **top-level agent** (WebSocket chat to browser clients via `useAgentChat`) and a **sub-agent** (RPC streaming from a parent agent via `chat()`). + +> **Experimental.** Think requires the `"experimental"` compatibility flag in your `wrangler.jsonc`. The API surface is stable but may evolve before graduating out of experimental. + +## Quick Start + +### Install + +```sh +npm install @cloudflare/think agents ai @cloudflare/shell zod workers-ai-provider +``` + +### Server + +```typescript +import { Think } from "@cloudflare/think"; +import { createWorkersAI } from "workers-ai-provider"; +import { routeAgentRequest } from "agents"; + +export class MyAgent extends Think { + getModel() { + return createWorkersAI({ binding: this.env.AI })( + "@cf/moonshotai/kimi-k2.5" + ); + } +} + +export default { + async fetch(request: Request, env: Env) { + return ( + (await routeAgentRequest(request, env)) || + new Response("Not found", { status: 404 }) + ); + } +} satisfies ExportedHandler; +``` + +That is it. Think handles the WebSocket chat protocol, message persistence, the agentic loop, message sanitization, stream resumption, client tool support, and workspace file tools. + +### Client + +```tsx +import { useAgent } from "agents/react"; +import { useAgentChat } from "@cloudflare/ai-chat/react"; + +function Chat() { + const agent = useAgent({ agent: "MyAgent" }); + const { messages, sendMessage, status } = useAgentChat({ agent }); + + return ( +
+ {messages.map((msg) => ( +
+ {msg.role}: + {msg.parts.map((part, i) => + part.type === "text" ? {part.text} : null + )} +
+ ))} + +
{ + e.preventDefault(); + const input = e.currentTarget.elements.namedItem( + "input" + ) as HTMLInputElement; + sendMessage({ text: input.value }); + input.value = ""; + }} + > + + +
+
+ ); +} +``` + +### wrangler.jsonc + +```jsonc +{ + "compatibility_date": "2026-01-28", + "compatibility_flags": ["nodejs_compat", "experimental"], + "ai": { "binding": "AI" }, + "durable_objects": { + "bindings": [{ "class_name": "MyAgent", "name": "MyAgent" }] + }, + "migrations": [{ "new_sqlite_classes": ["MyAgent"], "tag": "v1" }], + "main": "src/server.ts" +} +``` + +## Think vs AIChatAgent + +Both Think and [`AIChatAgent`](../chat-agents.md) extend `Agent` and speak the same `cf_agent_chat_*` WebSocket protocol. They serve different goals. + +**AIChatAgent** is a protocol adapter. You override `onChatMessage` and are responsible for calling `streamText`, wiring tools, converting messages, and returning a `Response`. AIChatAgent handles the plumbing — message persistence, streaming, abort, resume — but the LLM call is entirely your concern. + +**Think** is an opinionated framework. It makes decisions for you: `getModel()` returns the model, `getSystemPrompt()` or `configureSession()` sets the prompt, `getTools()` returns tools. The default `onChatMessage` runs the complete agentic loop. You override individual pieces, not the whole pipeline. + +| Concern | AIChatAgent | Think | +| ---------------------- | ---------------------------------------------------------------- | ------------------------------------------------------------------- | +| **Minimal subclass** | ~15 lines (wire `streamText` + tools + system prompt + response) | 3 lines (`getModel()` only) | +| **Storage** | Flat SQL table | Session: tree-structured messages, context blocks, compaction, FTS5 | +| **Regeneration** | Destructive (old response deleted) | Non-destructive branching (old responses preserved) | +| **Context management** | Manual | Context blocks with LLM-writable persistent memory | +| **Sub-agent RPC** | Not built in | `chat()` with `StreamCallback` | +| **Programmatic turns** | `saveMessages()` | `saveMessages()` + `continueLastTurn()` | +| **Compaction** | `maxPersistedMessages` (deletes oldest) | Non-destructive summaries via overlays | +| **Search** | Not available | FTS5 full-text search per-session and cross-session | + +### When to use AIChatAgent + +- You need full control over the LLM call (RAG, multi-model, custom streaming) +- You are migrating from AI SDK v4 (`autoTransformMessages` provides the bridge) +- You want the `Response` return type for HTTP middleware or testing +- You are building a simple chatbot with no memory requirements + +### When to use Think + +- You want to ship fast (3-line subclass with everything wired) +- You need persistent memory (context blocks the model can read and write) +- You need long conversations (non-destructive compaction) +- You need conversation search (FTS5) +- You are building a sub-agent system (parent-child RPC with streaming) +- You need proactive agents (programmatic turns from scheduled tasks or webhooks) + +## Configuration Overrides + +| Method / Property | Default | Description | +| ----------------------- | -------------------------------- | ------------------------------------------------------------------------------- | +| `getModel()` | throws | Return the `LanguageModel` to use | +| `getSystemPrompt()` | `"You are a helpful assistant."` | System prompt (fallback when no context blocks) | +| `getTools()` | `{}` | AI SDK `ToolSet` for the agentic loop | +| `maxSteps` | `10` | Max tool-call rounds per turn | +| `configureSession()` | identity | Add context blocks, compaction, search, skills — see [Sessions](../sessions.md) | +| `messageConcurrency` | `"queue"` | How overlapping submits behave — see [Client Tools](./client-tools.md) | +| `waitForMcpConnections` | `false` | Wait for MCP servers before inference | +| `chatRecovery` | `true` | Wrap turns in `runFiber` for durable execution | + +## Dynamic Configuration + +Think accepts a `Config` type parameter for per-instance typed configuration. Configuration is persisted in SQLite and survives hibernation and restarts. + +```typescript +type MyConfig = { modelTier: "fast" | "capable"; theme: string }; + +export class MyAgent extends Think { + getModel() { + const tier = this.getConfig()?.modelTier ?? "fast"; + const models = { + fast: "@cf/moonshotai/kimi-k2.5", + capable: "@cf/meta/llama-4-scout-17b-16e-instruct" + }; + return createWorkersAI({ binding: this.env.AI })(models[tier]); + } +} +``` + +| Method | Description | +| ----------------------------- | ------------------------------------------------------------- | +| `configure(config: Config)` | Persist a typed configuration object | +| `getConfig(): Config \| null` | Read the persisted configuration, or null if never configured | + +Expose configuration to the client via `@callable`: + +```typescript +import { callable } from "agents"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + @callable() + updateConfig(config: MyConfig) { + this.configure(config); + } +} +``` + +## Session Integration + +Think uses [Session](../sessions.md) for conversation storage. Override `configureSession` to add persistent memory, compaction, search, and skills: + +```typescript +import { Think, Session } from "@cloudflare/think"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + configureSession(session: Session) { + return session + .withContext("soul", { + provider: { get: async () => "You are a helpful coding assistant." } + }) + .withContext("memory", { + description: "Important facts learned during conversation.", + maxTokens: 2000 + }) + .withCachedPrompt(); + } +} +``` + +Think's `this.messages` getter reads directly from Session's tree-structured storage. Context blocks, compaction overlays, and search are all handled by Session. See the [Sessions documentation](../sessions.md) for the full API. + +## Package Exports + +| Export | Description | +| ------------------------------------ | ------------------------------------------------------------- | +| `@cloudflare/think` | `Think`, `Session`, `Workspace` — main class + re-exports | +| `@cloudflare/think/tools/workspace` | `createWorkspaceTools()` — for custom storage backends | +| `@cloudflare/think/tools/execute` | `createExecuteTool()` — sandboxed code execution via codemode | +| `@cloudflare/think/tools/extensions` | `createExtensionTools()` — LLM-driven extension loading | +| `@cloudflare/think/extensions` | `ExtensionManager`, `HostBridgeLoopback` — extension runtime | + +## Peer Dependencies + +| Package | Required | Notes | +| ---------------------- | -------- | ----------------------- | +| `agents` | yes | Cloudflare Agents SDK | +| `ai` | yes | Vercel AI SDK v6 | +| `zod` | yes | Schema validation (v4) | +| `@cloudflare/shell` | yes | Workspace filesystem | +| `@cloudflare/codemode` | optional | For `createExecuteTool` | + +## Docs + +- [Getting Started](./getting-started.md) — Build a Think agent step by step +- [Lifecycle Hooks](./lifecycle-hooks.md) — `beforeTurn`, `onStepFinish`, `onChunk`, `onChatResponse`, and more +- [Tools](./tools.md) — Workspace tools, code execution, extensions +- [Client Tools](./client-tools.md) — Browser-side tools, approvals, and concurrency +- [Sub-agents and Programmatic Turns](./sub-agents.md) — RPC streaming, `saveMessages`, recovery diff --git a/docs/think/lifecycle-hooks.md b/docs/think/lifecycle-hooks.md new file mode 100644 index 0000000000..f463d47230 --- /dev/null +++ b/docs/think/lifecycle-hooks.md @@ -0,0 +1,357 @@ +# Lifecycle Hooks + +Think owns the `streamText` call and provides hooks at each stage of the chat turn. Hooks fire on every turn regardless of entry path — WebSocket chat, sub-agent `chat()`, `saveMessages`, and auto-continuation after tool results. + +## Hook Summary + +| Hook | When it fires | Return | Async | +| --------------------------- | --------------------------------------------- | -------------------------- | ----- | +| `configureSession(session)` | Once during `onStart` | `Session` | yes | +| `beforeTurn(ctx)` | Before `streamText` | `TurnConfig` or void | yes | +| `beforeToolCall(ctx)` | When model calls a tool | `ToolCallDecision` or void | yes | +| `afterToolCall(ctx)` | After tool execution | void | yes | +| `onStepFinish(ctx)` | After each step completes | void | yes | +| `onChunk(ctx)` | Per streaming chunk | void | yes | +| `onChatResponse(result)` | After turn completes and message is persisted | void | yes | +| `onChatError(error)` | On error during a turn | error to propagate | no | + +## Execution Order + +For a turn with two tool calls: + +``` +configureSession() ← once at startup, not per-turn + │ +beforeTurn() ← inspect assembled context, override model/tools/prompt + │ + ┌── streamText ───────────────────────────────────┐ + │ onChunk() onChunk() onChunk() ... │ + │ │ │ + │ beforeToolCall() → tool executes │ + │ afterToolCall() │ + │ │ │ + │ onStepFinish() │ + │ │ │ + │ onChunk() onChunk() ... │ + │ │ │ + │ beforeToolCall() → tool executes │ + │ afterToolCall() │ + │ │ │ + │ onStepFinish() │ + └─────────────────────────────────────────────────┘ + │ +onChatResponse() ← message persisted, turn lock released +``` + +--- + +## configureSession + +Called once during Durable Object initialization (`onStart`). Configure the Session with context blocks, compaction, search, and skills. + +```typescript +configureSession(session: Session): Session | Promise +``` + +```typescript +import { Think, Session } from "@cloudflare/think"; +import { createCompactFunction } from "agents/experimental/memory/utils/compaction-helpers"; +import { generateText } from "ai"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + configureSession(session: Session) { + return session + .withContext("soul", { + provider: { get: async () => "You are a helpful coding assistant." } + }) + .withContext("memory", { + description: "Learned facts about the user.", + maxTokens: 1100 + }) + .onCompaction( + createCompactFunction({ + summarize: (prompt) => + generateText({ model: this.getModel(), prompt }).then((r) => r.text) + }) + ) + .compactAfter(100_000) + .withCachedPrompt(); + } +} +``` + +When `configureSession` adds context blocks, Think builds the system prompt from those blocks instead of using `getSystemPrompt()`. See the [Sessions documentation](../sessions.md) for the full API. + +--- + +## beforeTurn + +Called before `streamText`. Receives the fully assembled context — system prompt, converted messages, merged tools, and model. Return a `TurnConfig` to override any part, or void to accept defaults. + +```typescript +beforeTurn(ctx: TurnContext): TurnConfig | void | Promise +``` + +### TurnContext + +| Field | Type | Description | +| -------------- | ------------------------- | ------------------------------------------------------------------------ | +| `system` | `string` | Assembled system prompt (from context blocks or `getSystemPrompt()`) | +| `messages` | `ModelMessage[]` | Assembled model messages (truncated, pruned) | +| `tools` | `ToolSet` | Merged tool set (workspace + getTools + session + MCP + client + caller) | +| `model` | `LanguageModel` | The model from `getModel()` | +| `continuation` | `boolean` | Whether this is a continuation turn (auto-continue after tool result) | +| `body` | `Record` | Custom body fields from the client request | + +### TurnConfig + +All fields are optional. Return only what you want to change. + +| Field | Type | Description | +| ----------------- | ------------------------- | ------------------------------------ | +| `model` | `LanguageModel` | Override the model for this turn | +| `system` | `string` | Override the system prompt | +| `messages` | `ModelMessage[]` | Override the assembled messages | +| `tools` | `ToolSet` | Extra tools to merge (additive) | +| `activeTools` | `string[]` | Limit which tools the model can call | +| `toolChoice` | `ToolChoice` | Force a specific tool call | +| `maxSteps` | `number` | Override `maxSteps` for this turn | +| `providerOptions` | `Record` | Provider-specific options | + +### Examples + +Switch to a cheaper model for continuation turns: + +```typescript +beforeTurn(ctx: TurnContext) { + if (ctx.continuation) { + return { model: this.cheapModel }; + } +} +``` + +Restrict which tools the model can call: + +```typescript +beforeTurn(ctx: TurnContext) { + return { activeTools: ["read", "write", "getWeather"] }; +} +``` + +Add per-turn context from the client body: + +```typescript +beforeTurn(ctx: TurnContext) { + if (ctx.body?.selectedFile) { + return { + system: ctx.system + `\n\nUser is editing: ${ctx.body.selectedFile}` + }; + } +} +``` + +Override `maxSteps` based on conversation length: + +```typescript +beforeTurn(ctx: TurnContext) { + if (ctx.messages.length > 100) { + return { maxSteps: 3 }; + } +} +``` + +--- + +## beforeToolCall + +Called when the model produces a tool call. Only fires for server-side tools (tools with `execute`). Client tools are handled on the client. + +> **Current limitation:** `beforeToolCall` currently fires as an observation hook — after tool execution, via `onStepFinish` data. The `block` and `substitute` actions in `ToolCallDecision` are defined in the types but are not yet functional. The AI SDK's `streamText` does not expose a pre-execution interception point in the Workers runtime. For now, use this hook for logging and analytics. + +```typescript +beforeToolCall(ctx: ToolCallContext): ToolCallDecision | void | Promise +``` + +### ToolCallContext + +| Field | Type | Description | +| ---------- | ------------------------- | ----------------------------- | +| `toolName` | `string` | Name of the tool being called | +| `args` | `Record` | Arguments the model provided | + +### ToolCallDecision (future) + +When pre-execution interception becomes available, the return type will support three actions: + +| Action | Fields | Behavior | +| -------------- | ----------------- | -------------------------------------------------- | +| `"allow"` | `args?` | Execute the tool, optionally with modified args | +| `"block"` | `reason?` | Do not execute; return `reason` as the tool result | +| `"substitute"` | `result`, `args?` | Do not execute; return `result` as the tool result | + +### Example + +Log all tool calls: + +```typescript +beforeToolCall(ctx: ToolCallContext) { + console.log(`Tool called: ${ctx.toolName}`, ctx.args); +} +``` + +--- + +## afterToolCall + +Called after a tool executes (or a substitute result is provided by `beforeToolCall`). Does not fire when `beforeToolCall` blocks with no substitute. + +```typescript +afterToolCall(ctx: ToolCallResultContext): void | Promise +``` + +### ToolCallResultContext + +| Field | Type | Description | +| ---------- | ------------------------- | ------------------------------------------------------------------------ | +| `toolName` | `string` | Name of the tool that was called | +| `args` | `Record` | Arguments the tool was called with (may be modified by `beforeToolCall`) | +| `result` | `unknown` | The result returned by the tool | + +### Example + +Track tool usage: + +```typescript +afterToolCall(ctx: ToolCallResultContext) { + this.env.ANALYTICS.writeDataPoint({ + blobs: [ctx.toolName], + doubles: [JSON.stringify(ctx.result).length] + }); +} +``` + +--- + +## onStepFinish + +Called after each step completes in the agentic loop. A step is one `streamText` iteration — the model generates text, optionally calls tools, and the step ends. + +```typescript +onStepFinish(ctx: StepContext): void | Promise +``` + +### StepContext + +| Field | Type | Description | +| -------------- | ------------------------------------------ | --------------------------- | +| `stepType` | `"initial" \| "continue" \| "tool-result"` | Why the step ran | +| `text` | `string` | Text generated in this step | +| `toolCalls` | `unknown[]` | Tool calls made | +| `toolResults` | `unknown[]` | Tool results received | +| `finishReason` | `string` | Why the step ended | +| `usage` | `{ inputTokens, outputTokens }` | Token usage for this step | + +### Example + +Log step-level usage: + +```typescript +onStepFinish(ctx: StepContext) { + console.log( + `Step ${ctx.stepType}: ${ctx.usage.inputTokens}in/${ctx.usage.outputTokens}out` + ); +} +``` + +--- + +## onChunk + +Called for each streaming chunk. High-frequency — fires per token. Override for streaming analytics, progress indicators, or token counting. Observational only. + +```typescript +onChunk(ctx: ChunkContext): void | Promise +``` + +### ChunkContext + +| Field | Type | Description | +| ------- | --------- | ------------------------------------- | +| `chunk` | `unknown` | The chunk data from the AI SDK stream | + +--- + +## onChatResponse + +Called after a chat turn completes and the assistant message has been persisted. The turn lock is released before this hook runs, so it is safe to call `saveMessages` or other methods from inside. + +Fires for all turn completion paths: WebSocket, sub-agent RPC, `saveMessages`, and auto-continuation. + +```typescript +onChatResponse(result: ChatResponseResult): void | Promise +``` + +### ChatResponseResult + +| Field | Type | Description | +| -------------- | ------------------------------------- | ------------------------------------------ | +| `message` | `UIMessage` | The persisted assistant message | +| `requestId` | `string` | Unique ID for this turn | +| `continuation` | `boolean` | Whether this was a continuation turn | +| `status` | `"completed" \| "error" \| "aborted"` | How the turn ended | +| `error` | `string?` | Error message (when `status` is `"error"`) | + +### Examples + +Log turn completion: + +```typescript +onChatResponse(result: ChatResponseResult) { + if (result.status === "completed") { + console.log( + `Turn ${result.requestId} completed: ${result.message.parts.length} parts` + ); + } +} +``` + +Chain a follow-up turn: + +```typescript +async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed" && this.shouldFollowUp(result.message)) { + await this.saveMessages([{ + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Now summarize what you found." }] + }]); + } +} +``` + +--- + +## onChatError + +Called when an error occurs during a chat turn. Return the error to propagate it, or return a different error. + +```typescript +onChatError(error: unknown): unknown +``` + +The partial assistant message (if any) is persisted before this hook fires. + +### Example + +Log and transform errors: + +```typescript +onChatError(error: unknown) { + console.error("Chat turn failed:", error); + return new Error("Something went wrong. Please try again."); +} +``` diff --git a/docs/think/sub-agents.md b/docs/think/sub-agents.md new file mode 100644 index 0000000000..09c5905677 --- /dev/null +++ b/docs/think/sub-agents.md @@ -0,0 +1,339 @@ +# Sub-agents and Programmatic Turns + +Think works as both a top-level agent (WebSocket to browser) and a sub-agent (RPC from a parent agent). It also supports programmatic turns — injecting messages and triggering model turns without a WebSocket connection. + +## Sub-agent via chat() + +When used as a sub-agent, the `chat()` method runs a full turn (persist user message, run agentic loop, persist assistant response) and streams events via a callback. + +```typescript +async chat( + userMessage: string | UIMessage, + callback: StreamCallback, + options?: ChatOptions +): Promise +``` + +### StreamCallback + +```typescript +interface StreamCallback { + onEvent(json: string): void | Promise; + onDone(): void | Promise; + onError?(error: string): void | Promise; +} +``` + +| Method | When it fires | +| ---------------- | --------------------------------------------------------------- | +| `onEvent(json)` | For each streaming chunk (JSON-serialized UIMessageChunk) | +| `onDone()` | After the turn completes and the assistant message is persisted | +| `onError(error)` | On error during the turn (if not provided, the error is thrown) | + +### ChatOptions + +```typescript +interface ChatOptions { + signal?: AbortSignal; + tools?: ToolSet; +} +``` + +| Field | Description | +| -------- | ----------------------------------------------------------- | +| `signal` | `AbortSignal` to cancel the turn mid-stream | +| `tools` | Extra tools to merge for this turn (highest merge priority) | + +### Example: Parent agent calling a child + +```typescript +import { Think, Session } from "@cloudflare/think"; +import type { StreamCallback } from "@cloudflare/think"; + +export class ParentAgent extends Think { + getModel() { + /* ... */ + } + + async delegateToChild(task: string) { + const child = await this.subAgent(ChildAgent, "child-1"); + + const chunks: string[] = []; + await child.chat(task, { + onEvent: (json) => { + chunks.push(json); + // Optionally forward to a connected client + }, + onDone: () => { + console.log("Child completed"); + }, + onError: (error) => { + console.error("Child failed:", error); + } + }); + + return chunks; + } +} + +export class ChildAgent extends Think { + getModel() { + /* ... */ + } + + getSystemPrompt() { + return "You are a research assistant. Analyze data and report findings."; + } +} +``` + +### Passing a string vs UIMessage + +`chat()` accepts either a plain string or a `UIMessage`. A string is auto-wrapped: + +```typescript +// These are equivalent: +await child.chat("Analyze this data", callback); +await child.chat( + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Analyze this data" }] + }, + callback +); +``` + +### Passing extra tools + +The `tools` option adds tools for this turn only, with the highest merge priority: + +```typescript +await child.chat("Summarize the report", callback, { + tools: { + fetchReport: tool({ + description: "Fetch the report data", + inputSchema: z.object({}), + execute: async () => this.getReportData() + }) + } +}); +``` + +### Aborting a sub-agent turn + +Pass an `AbortSignal` to cancel mid-stream: + +```typescript +const controller = new AbortController(); + +setTimeout(() => controller.abort(), 30_000); + +await child.chat("Long analysis task", callback, { + signal: controller.signal +}); +``` + +When aborted, the partial assistant message is still persisted. + +--- + +## Programmatic Turns with saveMessages + +`saveMessages` injects messages and triggers a model turn without a WebSocket connection. Use for scheduled responses, webhook-triggered turns, proactive agents, or chaining from `onChatResponse`. + +```typescript +async saveMessages( + messages: UIMessage[] | ((current: UIMessage[]) => UIMessage[] | Promise) +): Promise +``` + +Returns `{ requestId, status }` where `status` is `"completed"` or `"skipped"`. + +### Static messages + +```typescript +await this.saveMessages([ + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Time for your daily summary." }] + } +]); +``` + +### Function form + +When multiple `saveMessages` calls queue up, the function form runs with the latest messages when the turn actually starts: + +```typescript +await this.saveMessages((current) => [ + ...current, + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Continue your analysis." }] + } +]); +``` + +### Scheduled responses + +Trigger a turn from a cron schedule: + +```typescript +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + async onScheduled() { + await this.saveMessages([ + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Generate the daily report." }] + } + ]); + } +} +``` + +### Chaining from onChatResponse + +Start a follow-up turn after the current one completes: + +```typescript +async onChatResponse(result: ChatResponseResult) { + if (result.status === "completed" && this.needsFollowUp(result.message)) { + await this.saveMessages([{ + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Now summarize what you found." }] + }]); + } +} +``` + +--- + +## continueLastTurn + +Resume the last assistant turn without injecting a new user message. Useful after tool results are received or after recovery from an interruption. + +```typescript +protected async continueLastTurn( + body?: Record +): Promise +``` + +Returns `{ requestId, status: "skipped" }` if the last message is not an assistant message. + +The optional `body` parameter overrides the stored body for this continuation. If omitted, the last body from the previous turn is used. + +--- + +## Chat Recovery + +Think can wrap chat turns in Durable Object fibers for durable execution. When a DO is evicted mid-turn, the turn can be recovered on restart. + +### Setup + +```typescript +export class MyAgent extends Think { + chatRecovery = true; + + getModel() { + /* ... */ + } +} +``` + +When `chatRecovery` is `true`, all four turn paths (WebSocket, auto-continuation, `saveMessages`, `continueLastTurn`) are wrapped in `runFiber`. + +### onChatRecovery + +When an interrupted chat fiber is detected after DO restart, Think calls the `onChatRecovery` hook: + +```typescript +onChatRecovery(ctx: ChatRecoveryContext): ChatRecoveryOptions | void +``` + +### ChatRecoveryContext + +| Field | Type | Description | +| ----------------- | -------------------------- | ----------------------------------------- | +| `streamId` | `string` | The stream ID of the interrupted turn | +| `requestId` | `string` | The request ID of the interrupted turn | +| `partialText` | `string` | Text generated before the interruption | +| `partialParts` | `MessagePart[]` | Parts accumulated before the interruption | +| `recoveryData` | `unknown \| null` | Data from `this.stash()` during the turn | +| `messages` | `UIMessage[]` | Current conversation history | +| `lastBody` | `Record?` | Body from the interrupted turn | +| `lastClientTools` | `ClientToolSchema[]?` | Client tools from the interrupted turn | + +### ChatRecoveryOptions + +| Field | Type | Description | +| ---------- | ---------- | ------------------------------------------------ | +| `persist` | `boolean?` | Whether to persist the partial assistant message | +| `continue` | `boolean?` | Whether to auto-continue with a new turn | + +### Example + +```typescript +export class MyAgent extends Think { + chatRecovery = true; + + getModel() { + /* ... */ + } + + onChatRecovery(ctx: ChatRecoveryContext) { + console.log( + `Recovering turn ${ctx.requestId}, partial: ${ctx.partialText.length} chars` + ); + return { + persist: true, + continue: true + }; + } +} +``` + +With `persist: true`, the partial message is saved. With `continue: true`, Think calls `continueLastTurn()` after the agent reaches a stable state. + +--- + +## Stability Detection + +Think provides methods to check if the agent is in a stable state — no pending tool results, no pending approvals, no active turns. + +### hasPendingInteraction + +```typescript +protected hasPendingInteraction(): boolean +``` + +Returns `true` if any assistant message has pending tool calls (tools without results or pending approvals). + +### waitUntilStable + +```typescript +protected async waitUntilStable(options?: { timeout?: number }): Promise +``` + +Returns a promise that resolves to `true` when the agent reaches a stable state, or `false` if the timeout is exceeded. + +```typescript +const stable = await this.waitUntilStable({ timeout: 30_000 }); +if (stable) { + await this.saveMessages([ + { + id: crypto.randomUUID(), + role: "user", + parts: [{ type: "text", text: "Now that you are done, summarize." }] + } + ]); +} +``` diff --git a/docs/think/tools.md b/docs/think/tools.md new file mode 100644 index 0000000000..a4f2256d9d --- /dev/null +++ b/docs/think/tools.md @@ -0,0 +1,491 @@ +# Tools + +Think provides built-in workspace file tools on every turn, plus integration points for custom tools, code execution, and dynamic extensions. + +## Tool Merge Order + +On every turn, Think merges tools from multiple sources. Later sources override earlier ones if names collide: + +1. **Workspace tools** — `read`, `write`, `edit`, `list`, `find`, `grep`, `delete` (built-in) +2. **`getTools()`** — your custom server-side tools +3. **Session tools** — `set_context`, `load_context`, `search_context` (from `configureSession`) +4. **Extension tools** — tools from loaded extensions (prefixed by extension name) +5. **MCP tools** — from connected MCP servers +6. **Client tools** — from the browser (see [Client Tools](./client-tools.md)) +7. **Caller tools** — from `chat()` options when used as a sub-agent + +## Built-in Workspace Tools + +Every Think agent gets `this.workspace` — a virtual filesystem backed by the Durable Object's SQLite storage. Workspace tools are automatically available to the model with no configuration. + +| Tool | Description | +| -------- | --------------------------------------------------------------------------- | +| `read` | Read a file's content | +| `write` | Write content to a file (creates parent directories) | +| `edit` | Apply a find-and-replace edit to an existing file (supports fuzzy matching) | +| `list` | List files and directories in a path | +| `find` | Find files matching a glob pattern | +| `grep` | Search file contents by regex or fixed string | +| `delete` | Delete a file or directory | + +### R2 Spillover + +By default, the workspace stores everything in SQLite. For large files, override `workspace` to add R2 spillover: + +```typescript +import { Think } from "@cloudflare/think"; +import { Workspace } from "@cloudflare/shell"; + +export class MyAgent extends Think { + override workspace = new Workspace({ + sql: this.ctx.storage.sql, + r2: this.env.R2, + name: () => this.name + }); + + getModel() { + /* ... */ + } +} +``` + +This requires an R2 bucket binding in `wrangler.jsonc`: + +```jsonc +{ + "r2_buckets": [{ "binding": "R2", "bucket_name": "agent-files" }] +} +``` + +## Custom Tools + +Override `getTools()` to add your own tools. These are standard AI SDK `tool()` definitions with Zod schemas: + +```typescript +import { Think } from "@cloudflare/think"; +import { tool } from "ai"; +import type { ToolSet } from "ai"; +import { z } from "zod"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + getTools(): ToolSet { + return { + getWeather: tool({ + description: "Get the current weather for a city", + inputSchema: z.object({ + city: z.string().describe("City name") + }), + execute: async ({ city }) => { + const res = await fetch( + `https://api.weather.com/v1/current?q=${city}&key=${this.env.WEATHER_KEY}` + ); + return res.json(); + } + }), + + calculate: tool({ + description: "Perform a math calculation", + inputSchema: z.object({ + a: z.number(), + b: z.number(), + operator: z.enum(["+", "-", "*", "/"]) + }), + execute: async ({ a, b, operator }) => { + const ops: Record number> = { + "+": (x, y) => x + y, + "-": (x, y) => x - y, + "*": (x, y) => x * y, + "/": (x, y) => x / y + }; + return { result: ops[operator](a, b) }; + } + }) + }; + } +} +``` + +Custom tools are merged with workspace tools automatically. If a custom tool has the same name as a workspace tool, the custom tool wins. + +### Tool Approval + +Tools can require user approval before execution using the `needsApproval` option: + +```typescript +getTools(): ToolSet { + return { + deleteFile: tool({ + description: "Delete a file from the system", + inputSchema: z.object({ path: z.string() }), + needsApproval: async ({ path }) => path.startsWith("/important/"), + execute: async ({ path }) => { + await this.workspace.rm(path); + return { deleted: path }; + } + }) + }; +} +``` + +When `needsApproval` returns `true`, the tool call is sent to the client for approval. The conversation pauses until the client responds with `CF_AGENT_TOOL_APPROVAL`. See [Client Tools](./client-tools.md) for the approval flow. + +### Per-turn Tool Overrides + +The `beforeTurn` hook can restrict or add tools for a specific turn: + +```typescript +beforeTurn(ctx: TurnContext) { + return { + activeTools: ["read", "write", "getWeather"], + tools: { emergencyTool: this.createEmergencyTool() } + }; +} +``` + +`activeTools` limits which tools the model can call. `tools` adds extra tools for this turn only (merged on top of existing tools). + +## MCP Tools + +Think inherits MCP client support from the `Agent` base class. MCP tools from connected servers are automatically merged into every turn. + +Set `waitForMcpConnections` to ensure MCP servers are connected before the inference loop runs: + +```typescript +export class MyAgent extends Think { + waitForMcpConnections = true; // default 10s timeout + // or: waitForMcpConnections = { timeout: 5000 }; + + getModel() { + /* ... */ + } +} +``` + +Add MCP servers programmatically or via `@callable` methods: + +```typescript +import { callable } from "agents"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + @callable() + async addServer(name: string, url: string) { + return await this.addMcpServer(name, url); + } + + @callable() + async removeServer(serverId: string) { + await this.removeMcpServer(serverId); + } +} +``` + +See [Connecting to MCP Servers](../mcp-client.md) for full MCP client documentation. + +## Code Execution Tool + +Let the LLM write and run JavaScript in a sandboxed Worker. Requires `@cloudflare/codemode` and a `worker_loaders` binding. + +```sh +npm install @cloudflare/codemode +``` + +```typescript +import { Think } from "@cloudflare/think"; +import { createExecuteTool } from "@cloudflare/think/tools/execute"; +import { createWorkspaceTools } from "@cloudflare/think/tools/workspace"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + getTools() { + return { + execute: createExecuteTool({ + tools: createWorkspaceTools(this.workspace), + loader: this.env.LOADER + }) + }; + } +} +``` + +Add the `worker_loaders` binding in `wrangler.jsonc`: + +```jsonc +{ + "worker_loaders": [{ "binding": "LOADER" }] +} +``` + +The sandbox has access to `codemode.*` tool calls. For richer filesystem access, pass a `state` backend: + +```typescript +import { createWorkspaceStateBackend } from "@cloudflare/shell"; + +createExecuteTool({ + tools: myDomainTools, + state: createWorkspaceStateBackend(this.workspace), + loader: this.env.LOADER +}); +// sandbox: codemode.myTool() AND state.readFile(), state.planEdits(), etc. +``` + +## Browser Tools + +Give your agent full access to the Chrome DevTools Protocol (CDP) for web page inspection, scraping, screenshots, and debugging. Requires `@cloudflare/codemode` and a Browser Rendering binding. + +```sh +npm install @cloudflare/codemode +``` + +```typescript +import { Think } from "@cloudflare/think"; +import { createBrowserTools } from "@cloudflare/think/tools/browser"; + +export class MyAgent extends Think { + getModel() { + /* ... */ + } + + getTools() { + return { + ...createBrowserTools({ + browser: this.env.BROWSER, + loader: this.env.LOADER + }) + }; + } +} +``` + +Add the Browser Rendering and Worker Loader bindings in `wrangler.jsonc`: + +```jsonc +{ + "browser": { "binding": "BROWSER" }, + "worker_loaders": [{ "binding": "LOADER" }] +} +``` + +This adds two tools to your agent: + +| Tool | Description | +| ----------------- | --------------------------------------------------------------------------------------- | +| `browser_search` | Query the CDP protocol spec to discover commands, events, and types | +| `browser_execute` | Run CDP commands against a live browser session (screenshots, DOM reads, JS evaluation) | + +Both tools use the code-mode pattern — the model writes JavaScript async arrow functions that run in a sandboxed Worker isolate. In `browser_search`, the sandbox has access to `spec.get()` which returns the full normalized CDP protocol. In `browser_execute`, the sandbox has access to `cdp.send()`, `cdp.attachToTarget()`, and debug log helpers. + +Each `browser_execute` call opens a fresh browser session and closes it when the code finishes. For page-scoped CDP commands (`Page.*`, `Runtime.*`, `DOM.*`), the model must create a target, attach to it, and pass the `sessionId`. + +### Combining with Other Tools + +Browser tools compose naturally with workspace tools, code execution, MCP, and extensions: + +```typescript +import { createBrowserTools } from "@cloudflare/think/tools/browser"; +import { createExecuteTool } from "@cloudflare/think/tools/execute"; + +export class ResearchAgent extends Think { + getModel() { + /* ... */ + } + + getTools() { + return { + // Browse the web + ...createBrowserTools({ + browser: this.env.BROWSER, + loader: this.env.LOADER + }), + // Run sandboxed code against workspace files + execute: createExecuteTool({ + tools: createWorkspaceTools(this.workspace), + loader: this.env.LOADER + }) + }; + } +} +``` + +### Custom CDP Endpoint + +To connect to a Chrome instance running outside of Browser Rendering (e.g. `chrome --remote-debugging-port=9222`), pass `cdpUrl` instead of `browser`: + +```typescript +createBrowserTools({ + cdpUrl: "http://localhost:9222", + loader: this.env.LOADER +}); +``` + +See [Browse the Web](../browse-the-web.md) for the full CDP helper API reference, security model, and limitations. + +## Extensions + +Extensions are dynamically loaded sandboxed Workers that add tools at runtime. The LLM can write extension source code, load it, and use the new tools on the next turn. + +### Setup + +Extensions require `extensionLoader` (a `worker_loaders` binding) and the `ExtensionManager`: + +```typescript +import { Think } from "@cloudflare/think"; +import { ExtensionManager } from "@cloudflare/think/extensions"; +import { createExtensionTools } from "@cloudflare/think/tools/extensions"; + +export class MyAgent extends Think { + extensionLoader = this.env.LOADER; + + getModel() { + /* ... */ + } +} +``` + +When `extensionLoader` is set, Think automatically creates an `ExtensionManager` and loads extensions from `getExtensions()`. + +### Static Extensions + +Define extensions that load at startup: + +```typescript +export class MyAgent extends Think { + extensionLoader = this.env.LOADER; + + getModel() { + /* ... */ + } + + getExtensions() { + return [ + { + manifest: { + name: "math", + version: "1.0.0", + permissions: { network: false } + }, + source: `({ + tools: { + add: { + description: "Add two numbers", + parameters: { + a: { type: "number" }, + b: { type: "number" } + }, + execute: async ({ a, b }) => ({ result: a + b }) + } + } + })` + } + ]; + } +} +``` + +Extension tools are namespaced — `math` extension with `add` tool becomes `math_add` in the model's tool set. + +### LLM-Driven Extensions + +Give the model `createExtensionTools` so it can load extensions dynamically: + +```typescript +import { createExtensionTools } from "@cloudflare/think/tools/extensions"; + +export class MyAgent extends Think { + extensionLoader = this.env.LOADER; + + getModel() { + /* ... */ + } + + getTools() { + return { + ...createExtensionTools({ manager: this.extensionManager! }), + ...this.extensionManager!.getTools() + }; + } +} +``` + +This gives the model two tools: + +- `load_extension` — load a new extension from JavaScript source +- `list_extensions` — list currently loaded extensions + +Loaded extensions persist across DO restarts via `extensionManager.restore()`. + +### Extension Context Blocks + +Extensions can declare context blocks in their manifest. These are automatically registered with the Session: + +```typescript +getExtensions() { + return [{ + manifest: { + name: "notes", + version: "1.0.0", + permissions: { network: false }, + context: [ + { label: "scratchpad", description: "Extension scratch space", maxTokens: 500 } + ] + }, + source: `({ tools: { /* ... */ } })` + }]; +} +``` + +The context block is registered as `notes_scratchpad` (namespaced by extension name). + +## Workspace Tools for Custom Backends + +The individual tool factories are exported for use with custom storage backends that implement the operations interfaces: + +```typescript +import { + createReadTool, + createWriteTool, + createEditTool, + createListTool, + createFindTool, + createGrepTool, + createDeleteTool +} from "@cloudflare/think/tools/workspace"; +import type { + ReadOperations, + WriteOperations, + EditOperations, + ListOperations, + FindOperations, + GrepOperations, + DeleteOperations +} from "@cloudflare/think/tools/workspace"; +``` + +Implement the operations interface for your storage backend: + +```typescript +const myReadOps: ReadOperations = { + readFile: async (path) => fetchFromMyStorage(path), + stat: async (path) => getFileInfo(path) +}; + +const readTool = createReadTool({ ops: myReadOps }); +``` + +Or create the full set from a Workspace: + +```typescript +import { createWorkspaceTools } from "@cloudflare/think/tools/workspace"; + +const tools = createWorkspaceTools(myCustomWorkspace); +``` diff --git a/docs/voice.md b/docs/voice.md new file mode 100644 index 0000000000..05d5ab04c3 --- /dev/null +++ b/docs/voice.md @@ -0,0 +1,661 @@ +# Voice Agents + +Build real-time voice agents with speech-to-text, text-to-speech, and conversation persistence. Audio streams over WebSocket — no SFU or meeting infrastructure required. + +## Overview + +`@cloudflare/voice` provides two server-side mixins and matching React hooks: + +| Export | Import | Purpose | +| ---------------- | -------------------------- | -------------------------------------------- | +| `withVoice` | `@cloudflare/voice` | Full voice agent: STT, LLM, TTS, persistence | +| `withVoiceInput` | `@cloudflare/voice` | STT-only: transcription without response | +| `useVoiceAgent` | `@cloudflare/voice/react` | React hook for `withVoice` agents | +| `useVoiceInput` | `@cloudflare/voice/react` | React hook for `withVoiceInput` agents | +| `VoiceClient` | `@cloudflare/voice/client` | Framework-agnostic client | + +Built on Cloudflare Durable Objects, you get: + +- **Real-time audio** — mic audio streams as binary WebSocket frames, TTS audio streams back +- **Automatic conversation persistence** — messages stored in SQLite, survive restarts +- **Streaming TTS** — LLM tokens are sentence-chunked and synthesized concurrently +- **Interruption handling** — user speech during playback cancels the current response +- **Continuous STT** — per-call transcriber session, model handles turn detection +- **Pipeline hooks** — intercept and transform text at every stage + +> **Experimental.** This API is under active development and will break between releases. Pin your version. + +## Quick Start + +### Install + +```sh +npm install @cloudflare/voice agents +``` + +### Server + +```typescript +import { Agent } from "agents"; +import { + withVoice, + WorkersAIFluxSTT, + WorkersAITTS, + type VoiceTurnContext +} from "@cloudflare/voice"; + +const VoiceAgent = withVoice(Agent); + +export class MyAgent extends VoiceAgent { + transcriber = new WorkersAIFluxSTT(this.env.AI); + tts = new WorkersAITTS(this.env.AI); + + async onTurn(transcript: string, context: VoiceTurnContext) { + return "Hello! I heard you say: " + transcript; + } +} +``` + +### Client (React) + +```tsx +import { useVoiceAgent } from "@cloudflare/voice/react"; + +function VoiceUI() { + const { + status, + transcript, + interimTranscript, + audioLevel, + isMuted, + startCall, + endCall, + toggleMute + } = useVoiceAgent({ agent: "MyAgent" }); + + return ( +
+

Status: {status}

+ + + + + + {interimTranscript && ( +

+ {interimTranscript} +

+ )} + + {transcript.map((msg, i) => ( +

+ {msg.role}: {msg.text} +

+ ))} +
+ ); +} +``` + +### Wrangler Config + +```jsonc +// wrangler.jsonc +{ + "ai": { "binding": "AI" }, + "durable_objects": { + "bindings": [{ "name": "MyAgent", "class_name": "MyAgent" }] + }, + "migrations": [{ "tag": "v1", "new_sqlite_classes": ["MyAgent"] }] +} +``` + +## How It Works + +``` +Browser Durable Object (withVoice) +┌──────────┐ ┌──────────────────────────┐ +│ Mic │ binary PCM (16kHz) │ Transcriber session │ +│ │ ──────────────────────► │ (per-call, continuous) │ +│ │ │ ↓ model detects turn │ +│ │ JSON: transcript │ onTurn() → your LLM code │ +│ │ ◄────────────────────── │ ↓ (sentence chunking) │ +│ │ binary: audio │ TTS │ +│ Speaker │ ◄────────────────────── │ │ +└──────────┘ └──────────────────────────┘ +``` + +1. The client captures mic audio and sends it as binary WebSocket frames (16kHz mono 16-bit PCM) +2. Audio streams continuously to the transcriber session (created at `start_call`, lives for the entire call) +3. The STT model detects when the user finishes an utterance and fires `onUtterance` +4. Your `onTurn()` method runs — typically an LLM call +5. The response is sentence-chunked and synthesized via TTS +6. Audio streams back to the client for playback + +## Server API: `withVoice` + +`withVoice(Agent)` adds the full voice pipeline to an Agent class. + +### Providers + +Set providers as class properties. Class field initializers run after `super()`, so `this.env` is available. + +| Property | Type | Required | Description | +| ------------- | ------------- | -------- | -------------------------------- | +| `transcriber` | `Transcriber` | Yes | Continuous per-call STT provider | +| `tts` | `TTSProvider` | Yes | Text-to-speech | + +```typescript +import { withVoice, WorkersAIFluxSTT, WorkersAITTS } from "@cloudflare/voice"; + +const VoiceAgent = withVoice(Agent); + +export class MyAgent extends VoiceAgent { + transcriber = new WorkersAIFluxSTT(this.env.AI); + tts = new WorkersAITTS(this.env.AI); +} +``` + +For runtime model switching (e.g. Flux vs Nova 3 dropdown), override `createTranscriber`: + +```typescript +export class MyAgent extends VoiceAgent { + tts = new WorkersAITTS(this.env.AI); + + createTranscriber(connection: Connection): Transcriber { + return new WorkersAIFluxSTT(this.env.AI); + } +} +``` + +### `onTurn(transcript, context)` + +**Required.** Called when the user finishes speaking and the transcript is ready. + +Return a `string`, `AsyncIterable`, or `ReadableStream` for streaming responses: + +**Simple response:** + +```typescript +async onTurn(transcript: string, context: VoiceTurnContext) { + return "You said: " + transcript; +} +``` + +**Streaming response (recommended for LLM):** + +```typescript +import { streamText, convertToModelMessages } from "ai"; +import { createWorkersAI } from "workers-ai-provider"; + +async onTurn(transcript: string, context: VoiceTurnContext) { + const workersai = createWorkersAI({ binding: this.env.AI }); + + const result = streamText({ + model: workersai("@cf/moonshotai/kimi-k2.5"), + system: "You are a helpful voice assistant. Keep responses concise.", + messages: [ + ...context.messages.map(m => ({ + role: m.role as "user" | "assistant", + content: m.content + })), + { role: "user", content: transcript } + ], + abortSignal: context.signal + }); + + return result.textStream; +} +``` + +The `context` object provides: + +| Field | Type | Description | +| ------------ | ------------------------------------------ | ---------------------------------- | +| `connection` | `Connection` | The WebSocket connection | +| `messages` | `Array<{ role: string; content: string }>` | Conversation history from SQLite | +| `signal` | `AbortSignal` | Aborted on interrupt or disconnect | + +### Lifecycle Hooks + +| Method | Description | +| ----------------------------- | ------------------------------------------- | +| `beforeCallStart(connection)` | Return `false` to reject the call | +| `onCallStart(connection)` | Called after a call is accepted | +| `onCallEnd(connection)` | Called when a call ends | +| `onInterrupt(connection)` | Called when user interrupts during playback | + +### Pipeline Hooks + +Intercept and transform data at each pipeline stage. Return `null` to skip the current utterance. + +| Method | Receives | Can skip? | +| ------------------------------------------ | --------------- | --------- | +| `afterTranscribe(transcript, connection)` | STT text | Yes | +| `beforeSynthesize(text, connection)` | Text before TTS | Yes | +| `afterSynthesize(audio, text, connection)` | Audio after TTS | Yes | + +```typescript +export class MyAgent extends VoiceAgent { + // Filter out short/noise transcripts + afterTranscribe(transcript: string, connection: Connection) { + if (transcript.length < 3) return null; // skip + return transcript; + } + + // Add SSML or modify text before TTS + beforeSynthesize(text: string, connection: Connection) { + return text.replace(/\bAI\b/g, "A.I."); // improve pronunciation + } +} +``` + +### Convenience Methods + +| Method | Description | +| -------------------------- | -------------------------------------------- | +| `speak(connection, text)` | Synthesize and send audio to one connection | +| `speakAll(text)` | Synthesize and send audio to all connections | +| `forceEndCall(connection)` | Programmatically end a call | +| `saveMessage(role, text)` | Persist a message to conversation history | +| `getConversationHistory()` | Retrieve conversation history from SQLite | + +### Configuration Options + +Pass options to `withVoice()` as the second argument: + +```typescript +const VoiceAgent = withVoice(Agent, { + historyLimit: 20, // Max messages loaded for context (default: 20) + audioFormat: "mp3", // Audio format sent to client (default: "mp3") + maxMessageCount: 1000 // Max messages in SQLite (default: 1000) +}); +``` + +## Server API: `withVoiceInput` + +`withVoiceInput(Agent)` adds STT-only voice input — no TTS, no LLM, no response generation. Use this for dictation, search-by-voice, or any UI where you need speech-to-text without a conversational agent. + +```typescript +import { Agent } from "agents"; +import { withVoiceInput, WorkersAINova3STT } from "@cloudflare/voice"; + +const InputAgent = withVoiceInput(Agent); + +export class DictationAgent extends InputAgent { + transcriber = new WorkersAINova3STT(this.env.AI); + + onTranscript(text: string, connection: Connection) { + console.log("User said:", text); + } +} +``` + +### `onTranscript(text, connection)` + +Called after each utterance is transcribed. Override this to process the transcript. + +### Hooks + +`withVoiceInput` supports the same lifecycle hooks as `withVoice`: + +- `beforeCallStart(connection)` — return `false` to reject +- `onCallStart(connection)`, `onCallEnd(connection)`, `onInterrupt(connection)` +- `createTranscriber(connection)` — override for runtime model switching +- `afterTranscribe(transcript, connection)` — filter or transform transcripts + +It does **not** have TTS hooks (`beforeSynthesize`, `afterSynthesize`) or `onTurn`. + +## Client API: React Hooks + +### `useVoiceAgent` + +Wraps `VoiceClient` for `withVoice` agents. Manages connection, mic capture, playback, silence detection, and interrupt detection. + +```tsx +import { useVoiceAgent } from "@cloudflare/voice/react"; + +const { + status, // "idle" | "listening" | "thinking" | "speaking" + transcript, // TranscriptMessage[] — conversation history + interimTranscript, // string | null — real-time partial transcript + metrics, // VoicePipelineMetrics | null + audioLevel, // number (0–1) — current mic RMS level + isMuted, // boolean + connected, // boolean — WebSocket connected + error, // string | null + startCall, // () => Promise + endCall, // () => void + toggleMute, // () => void + sendText, // (text: string) => void — bypass STT + sendJSON, // (data: Record) => void + lastCustomMessage // unknown — last non-voice message from server +} = useVoiceAgent({ + agent: "MyAgent", // Required: Durable Object class name + name: "default", // Instance name (default: "default") + host: window.location.host // Host to connect to +}); +``` + +#### Tuning Options + +| Option | Type | Default | Description | +| -------------------- | -------- | ------- | ------------------------------------------------ | +| `silenceThreshold` | `number` | `0.04` | RMS below this is silence | +| `silenceDurationMs` | `number` | `500` | Silence duration before `end_of_speech` (ms) | +| `interruptThreshold` | `number` | `0.05` | RMS to detect speech during playback | +| `interruptChunks` | `number` | `2` | Consecutive high-RMS chunks to trigger interrupt | + +Changing tuning options triggers a client reconnect (the connection key includes them). + +### `useVoiceInput` + +Lightweight hook for dictation / voice-to-text. Accumulates user transcripts into a single string. + +```tsx +import { useVoiceInput } from "@cloudflare/voice/react"; + +function Dictation() { + const { + transcript, // string — accumulated text from all utterances + interimTranscript, // string | null — current partial transcript + isListening, // boolean + audioLevel, // number (0–1) + isMuted, // boolean + error, // string | null + start, // () => Promise + stop, // () => void + toggleMute, // () => void + clear // () => void — clear accumulated transcript + } = useVoiceInput({ agent: "DictationAgent" }); + + return ( +
+