From a5fecc907289a231984bb92f013a736383b5074b Mon Sep 17 00:00:00 2001 From: Ronit Date: Sun, 17 May 2026 16:18:08 +0530 Subject: [PATCH 01/13] docs(sync-agent): design spec for dynamic per-integration sync engines MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Master spec for a 3-PR stack (F1 foundation primitives, F2 sync agent + Dropbox e2e, F3 polling + additional integrations). Covers: queue-triggered sync agent runs on every Pipedream connection; agent triages into a full sync-engine branch (file systems, databases, ticket trackers) or an api-only skill-writing branch (write-only APIs, broadcasters); per-sync Cloudflare DO facets for handler isolation with R2 json mirror as the main-agent-facing index layer and facet SQLite for handler-internal state (dedup, raw_events for probe loop). Central alarm scheduler in the supervisor — facet setAlarm empirically validated as unsupported by Cloudflare's runtime (cf-facet-alarm-test, 2026-05-17). Mirror = index, not content (Prime Directive) — agents enumerate structural primitives only; content stays at the source and is fetched on demand via exec_code. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../specs/2026-05-17-sync-agent-design.md | 1132 +++++++++++++++++ 1 file changed, 1132 insertions(+) create mode 100644 docs/superpowers/specs/2026-05-17-sync-agent-design.md diff --git a/docs/superpowers/specs/2026-05-17-sync-agent-design.md b/docs/superpowers/specs/2026-05-17-sync-agent-design.md new file mode 100644 index 0000000..928793b --- /dev/null +++ b/docs/superpowers/specs/2026-05-17-sync-agent-design.md @@ -0,0 +1,1132 @@ +# Sync Agent — Dynamic Per-Integration Sync Engines + +**Date:** 2026-05-17 +**Status:** Draft for review +**Owners:** Ronit +**Stack:** 3 PRs (PR-F1, PR-F2, PR-F3), branched off `feat/sync-agent-design` + +--- + +## 1. Goal + +When a user connects a third-party integration via the existing `/connect` flow, the connection event flows: client detection → kernel API → **`integration-sync` Cloudflare Queue** → queue consumer → `kernel.runSyncAgent(integration, account_id)`. That RPC spins up a **sync agent run** (sauna-style sub-agent, no thread binding) which: + +1. **Triages the integration.** Pipedream exposes 3,000+ integrations and most are API-wrappers with no syncable surface (LinkedIn, WhatsApp, MailChimp, Twilio, OpenAI, Sendgrid, Stripe-write, ...). Only a small subset have a navigable structure worth mirroring (file systems: Dropbox/Drive/OneDrive; databases: Airtable/Notion-DB; ticket trackers: Linear/GitHub/Jira; calendars: Google Calendar; wikis: Notion-pages/Confluence; messaging: Slack). The sync agent's first job is deciding which bucket this integration falls into. + +2. **Writes `skills/integrations/.md` always.** Every connected integration gets a skill file. This is the universal contract — the main agent reads it to learn how to use the integration. The skill comes in two flavors: + - **Sync-engine flavor** (for sync-able integrations): explains the mirror's shape, query patterns over the mirror, and `exec_code` snippets for fetching content / writing upstream. + - **API-only flavor** (for everything else): explains the API surface with `exec_code` snippets. No mirror, no sync engine. + +3. **Optionally installs a sync engine** — handler JS code, mirror json shape, webhook (or polling) wiring — but only when triage said it's worth it. Many integrations get just a skill and exit; no `sync_engine` row, no facet, no mirror, no webhook. + +4. Returns control. The main agent from then on uses the integration by reading its skill: for sync-able integrations it grep-reads the mirror for navigation then `exec_code`s for content/writes; for API-only integrations it just `exec_code`s. + +The user-facing demo is the "level-2 magic" — five minutes after the user connects Dropbox in the UI, the main agent correctly answers "where are my tax filings?" and "create a one-pager for portfolio company X under the right folder" with no hand-holding. **No `/sync` command needed to start the sync.** The connection itself is the trigger. + +For an API-only integration like MailChimp, the same trigger fires the same sync agent, but the agent finishes in 30 seconds with a 40-line skill that says "MailChimp is a broadcast-mail API; here's how to send campaigns via `exec_code` with `X-Pd-App: mailchimp`." No engine, no mirror. The main agent reads this skill when the user says "send a campaign to my list" and proceeds. + +### 1.1 Mirror = index, not content (load-bearing principle — when there IS a mirror) + +This principle applies only inside the **sync-engine branch** of the triage. API-only integrations don't have a mirror; the principle is moot for them. For the integrations that DO get a sync engine, the principle is load-bearing: + +**The mirror is a navigation/lookup index, not a backup of the user's data.** Mirroring full content (file bytes, page bodies, issue descriptions, every Airtable record) defeats the entire architecture — facet SQLite + R2 mirrors aren't sized to hold gigabytes of user content, and copying it would create freshness + privacy problems for no real benefit. Content is cheap to fetch on demand once you know what to ask for; "knowing what to ask for" is the expensive part, and that's what we mirror. + +The sync agent's job is to think like a senior engineer designing a sync engine for that specific integration: + +> "If I'm wiring up an Airtable sync, I do NOT mirror every record. I mirror the user's **bases, tables, fields, and field-type metadata**. Then when the user asks 'find the Series-A companies', the main agent reads the mirror to know there's an `Investor Pipeline` base with a `Portfolio Companies` table that has a `Stage` field of type `singleSelect`, and writes one `exec_code` call to Airtable's `filterByFormula` API. Result: instant answer, no record duplication, mirror stays small." + +Per-integration heuristics the agent should apply: + +| Integration | Mirror (index layer) | Fetched on-demand via `exec_code` (content layer) | +|---|---|---| +| **Dropbox** | folder tree: paths, names, ids, types, sizes, modified-at, mime | file bytes (download endpoint) | +| **Airtable** | bases + tables + fields (schemas + select options) | records (rows themselves) | +| **Linear** | teams + projects + issue headers (id, title, state, priority, assignee, updated_at, labels) | issue bodies, comments, attachments | +| **GitHub** | repos + open issue headers + open PR headers (title, labels, state, updated_at) | issue bodies, comments, file contents, commit history | +| **Notion** | page tree (id, title, parent, type, last_edited_at) | page bodies / block contents | +| **Google Drive** | file tree (id, path-or-parents, name, mime, size, modified-at) | file contents | +| **Google Calendar** | calendars + recent event headers (id, summary, start, end, attendees) | full event details, descriptions, attachments | +| **Slack** | workspaces + channels + recent message headers (id, channel, user, ts) | message bodies, threads, files | + +This principle gets baked into the sync agent's system prompt as a top-level directive (§3.10.3, Phase 2 — Schema design). If the agent starts to mirror content, it has misread the task. + +This is sauna-shaped, with two deliberate departures and one inherited primitive: + +- **Inherited from sauna: per-sync Cloudflare DO Facets.** Each sync runs inside its own DO facet (`ctx.facets.get(sync.id, ...)`). The facet has isolated SQLite (handler scratch space — dedup tables, ephemeral state), receives capabilities from the Kernel DO supervisor via RPC, and self-schedules its own renewal/poll alarms. Lifecycle primitives (`ctx.facets.abort()` on update, `ctx.facets.delete()` on delete) map cleanly to `update_sync` / `delete_sync`. Facets are not separately billed. +- **Departure 1: No `ctx.agent` in handlers.** Sync handlers stay inline JS — no runtime LLM reasoning. +- **Departure 2: Mirror is R2 json, not facet SQLite.** Per user direction. The main agent grep-reads `integrations//struct.json` with the existing `read_file` tool — no new query primitive. The facet's SQLite stays available for handler scratch (cursors, dedup) but is not the main-agent-facing data layer. +- **Departure 3: No `api_hosts` allowlist.** Every fetch from the handler runtime and from `exec_code` goes through the Pipedream proxy, which enforces the `X-Pd-App` slug ↔ host pairing itself. No additional egress gate. + +--- + +## 2. Non-goals (v1) + +- **Mirroring content** (file bytes, page bodies, issue descriptions, Airtable record rows, Slack message bodies, etc.). See §1.1 — the mirror is strictly the navigation/index layer; content is fetched on demand via `exec_code`. +- Real-time bidirectional sync with read-your-writes consistency. Writes go through `exec_code`; the mirror catches up via webhook with seconds of lag. Eventual consistency only. +- Migrations of pre-existing sync mirrors when the agent changes the json shape. `update_sync` replaces the mirror wholesale; if the shape changes incompatibly, the agent does a backfill into the new shape. +- Multi-tenant Pipedream apps for integrations without programmatic per-user webhook registration (Notion-class). Pipedream's webhook-bus solution is **research material for PR-F3**; v1 assumes Dropbox-class (programmatic per-user-ish — see §3.2) or stable-webhook-class (Linear, GitHub) only. +- Streaming UI surface for sync-agent progress. Task panel shows a single task row "Sync dropbox: " updated coarsely. +- A general-purpose schema or graph database for cross-integration JOIN. The mirror is one json per integration. JOINs happen in the model's head if at all. + +--- + +## 3. Architecture (traced backwards from the main agent) + +### 3.1 The main agent runtime, after one or more integrations have been connected + +The main agent does NOT receive a new injected block for integrations. It uses the existing `` block (lists Pipedream slugs the user has connected — already injected today) plus a single new line in the agent's system prompt teaching the convention: + +> For any integration listed in ``, the file `skills/integrations/.md` (if it exists in R2) contains usage guidance — read it before invoking that integration. The skill explains whether a local mirror exists, how to query it, and how to call the upstream API via `exec_code`. + +The main agent's runtime flow for a question that touches a connected integration: + +1. **Sees** the integration's slug in ``. +2. **Reads** `skills/integrations/.md` via the existing `read_file` tool. +3. **Branches** on what the skill says: + - Sync-engine skill: skill points to `integrations//struct.json` and describes its shape. Agent `read_file`s the mirror, greps inline, finds what it needs, then `exec_code`s the upstream API (with `X-Pd-App: `) for actual content / writes. + - API-only skill: skill describes the API surface only. Agent goes straight to `exec_code`. +4. **No per-integration tool wrappers exist** — `exec_code` is the universal surface for both reads and writes against the upstream. + +Why no injected block: 3,000+ Pipedream integrations means most users will end up with many connected slugs. Pre-injecting metadata for all of them is context bloat. The skill file *is* the authoritative description and the agent loads it lazily, only when it's relevant to the current turn. + +### 3.2 The three webhook archetypes + +The sync agent must support all three: + +**Archetype A — Stable programmatic webhook (Linear, GitHub on a repo, Slack events API).** Webhook URL is registered per-user via the integration's API using the user's OAuth token (Pipedream-proxied). URL never expires. Payload carries the full event content. Handler parses and updates the mirror. + +**Archetype B — Expiring channel + empty ping + cursor-fetch (Google Drive, Google Calendar, Dropbox).** Webhook URL is registered per-user. Some channels expire (Google: hours to days); some persist (Dropbox: forever, but limited to one per app). Notifications are *empty* — they say "something changed for this account" but not what. The handler must call the integration's delta endpoint with a saved cursor to fetch the actual diff. For expiring channels, a scheduler must renew before expiry. + +**Archetype C — No programmatic per-user webhook (Notion at the time of writing).** The app has a single webhook URL configured in the developer console; all users fire to it. Pipedream may offer a webhook-bus solution that demultiplexes by account_id — **this is research material for PR-F3**. Fallback is polling: handler exports `pollOnce(ctx)`, DO alarm fires it on an interval. + +v1 (PR-F1 + PR-F2) targets archetype B with Dropbox as the smoke integration. Archetypes A and C are explicitly covered in PR-F3 with additional smoke integrations. + +### 3.3 Data model — `sync_engine` table + +Populated **only when the sync agent decided to install a real sync engine** (sync-able integrations). API-only integrations leave this table untouched — their entire trace is the skill md in R2. Lives in the Kernel DO's SQLite via Drizzle. Schema (in `packages/models/src/schema/sync-engine.ts`): + +```ts +export const syncEngine = sqliteTable("sync_engine", { + id: text("id").primaryKey(), // ULID + integration: text("integration").notNull(), // "dropbox" — Pipedream app slug + accountId: text("account_id").notNull(), // Pipedream account ref + status: text("status", { enum: STATUSES }).notNull().default("discovering"), + strategy: text("strategy", { enum: STRATEGIES }).notNull(), + // Authoring artifacts (sync-agent-produced) + handlerJs: text("handler_js").notNull(), // ES module source string + schemaDdl: text("schema_ddl").notNull().default(""), // SQL DDL for the facet's own SQLite (dedup tables, raw_events for probe loop, cursor state, etc.). Empty string allowed for handlers that don't need any SQL state. Run by the supervisor at facet bootstrap. + schemaDoc: text("schema_doc").notNull(), // free-form prose describing the mirror json shape (R2) + skillPath: text("skill_path").notNull(), // "skills/integrations/dropbox.md" + mirrorPath: text("mirror_path").notNull(), // "integrations/dropbox/struct.json" + // Webhook state + webhookId: text("webhook_id"), // upstream's id for the registered webhook (for unregister) + webhookSecret: text("webhook_secret"), // for verifySignature + webhookExpiresAt: integer("webhook_expires_at", { mode: "timestamp" }), // null for non-expiring + renewalIntervalSec: integer("renewal_interval_sec"), // null for non-expiring + // Polling state + pollIntervalSec: integer("poll_interval_sec"), // null when strategy != "poll" + lastPolledAt: integer("last_polled_at", { mode: "timestamp" }), + // Bookkeeping + version: integer("version").notNull().default(1), // bumped on update_sync; cache key for Worker Loader + lastSyncedAt: integer("last_synced_at", { mode: "timestamp" }), + errorText: text("error_text"), // last failure summary, cleared on recovery + createdAt: integer({ mode: "timestamp" }).$defaultFn(() => new Date()).notNull(), + updatedAt: integer({ mode: "timestamp" }).$defaultFn(() => new Date()).$onUpdate(() => new Date()).notNull(), +}, (t) => [ + uniqueIndex("sync_engine_account_idx").on(t.integration, t.accountId), + index("sync_engine_status_idx").on(t.status), + index("sync_engine_renewal_idx").on(t.webhookExpiresAt), +]); + +export const STATUSES = ["discovering", "active", "errored", "deleted"] as const; +export const STRATEGIES = ["webhook_stable", "webhook_channel", "poll"] as const; +``` + +Plus a small **error-only** sidecar table — log failures, not successes. Success is implicit via `sync_engine.last_synced_at` advancing. Matches sauna's `sync_errors` shape: + +```ts +export const syncError = sqliteTable("sync_errors", { + id: integer("id").primaryKey({ autoIncrement: true }), + syncId: text("sync_id").notNull().references(() => syncEngine.id, { onDelete: "cascade" }), + occurredAt: integer("occurred_at", { mode: "timestamp" }).$defaultFn(() => new Date()).notNull(), + payloadPreview: text("payload_preview"), // truncated webhook body (or null for alarm-fired failures) + errorMessage: text("error_message").notNull(), + stackTrace: text("stack_trace"), + handlerVersion: integer("handler_version").notNull(), // sync_engine.version at the time of the throw +}, (t) => [ + index("sync_errors_sync_id_occurred_at_idx").on(t.syncId, t.occurredAt.desc()), +]); +``` + +Only inserted when something throws: +- `verifySignature` rejects → row with `error_message = "signature verification rejected"` + `payload_preview` + `handler_version`. +- `handleWebhook` throws → row with the throw's message + stack + payload. +- `renewWebhook` / `pollOnce` throws on alarm fire → row, `payloadPreview` null. + +DESC index on `(sync_id, occurred_at)` supports `SELECT ... ORDER BY occurred_at DESC LIMIT N` efficiently — that's the `get_sync_errors` read pattern. Cascade-deletes when the parent sync row is removed. + +The cursor (when an integration needs one) lives **inside the mirror json itself** under a reserved `_cursor` field. Reasoning: simpler than a separate column or sidecar file, atomic with the mirror write, and main-agent code that greps the mirror can simply ignore `_cursor`. + +### 3.4 The mirror — `integrations//struct.json` in R2 + +Path: `integrations//struct.json` (per deployment — the Kernel DO is single-tenant). + +Shape: **agent-authored at `create_sync_engine` time**, free-form within the constraint that it's a valid json document AND adheres to the index-not-content principle from §1.1. The agent's job is to pick the smallest shape that makes the integration efficiently navigable + queryable. Concretely: + +**Dropbox (file tree integration) — agent will naturally land on:** + +```json +{ + "_cursor": "AAH9p3...", + "_updated_at": "2026-05-17T18:31:01Z", + "entries": [ + { "id": "id:abc", "path": "/Personal/Tax/2024-form-1040.pdf", "name": "2024-form-1040.pdf", "type": "file", "size": 142331, "modified": "2026-04-15T03:21:00Z", "mime": "application/pdf" }, + { "id": "id:def", "path": "/Fund/Dimension-I/Portfolio/AcmeCorp/one-pager.md", "name": "one-pager.md", "type": "file", "size": 3104, "modified": "2026-05-12T11:02:00Z", "mime": "text/markdown" }, + ... + ] +} +``` + +`schema_doc`: +> Flat entries array. Each entry: `{id, path, name, type, size, modified, mime}`. `type` is "file" or "folder". Sorted by `path` ascending. `_cursor` is Dropbox's list_folder/continue cursor — handler reads, fetches deltas, applies, saves new cursor. File contents NOT mirrored; download via `https://content.dropboxapi.com/2/files/download` with `X-Pd-App: dropbox` from `exec_code`. + +**Airtable (database integration) — different shape entirely:** + +```json +{ + "_cursor": "...", + "_updated_at": "2026-05-17T18:31:01Z", + "bases": [ + { "id": "appXyz", "name": "Investor Pipeline" } + ], + "tables": [ + { + "base_id": "appXyz", + "id": "tblAbc", + "name": "Portfolio Companies", + "primary_field_id": "fldName", + "fields": [ + { "id": "fldName", "name": "Company Name", "type": "singleLineText" }, + { "id": "fldStage", "name": "Stage", "type": "singleSelect", "options": ["Seed", "Series A", "Series B", "Series C"] }, + { "id": "fldCheck", "name": "Check Size", "type": "currency" }, + { "id": "fldNotes", "name": "Notes", "type": "longText" } + ] + }, + { + "base_id": "appXyz", + "id": "tblDef", + "name": "Deal Flow", + "fields": [ ... ] + } + ] +} +``` + +`schema_doc`: +> Index of the user's Airtable workspace. `bases` lists workspaces, `tables` lists tables within each base with their full field schema (incl. select options for enums). Records NOT mirrored — main agent fetches them via Airtable API on demand using the field IDs from this index. + +**The principle in action:** the agent does NOT enumerate every record in the Airtable. It enumerates bases + tables + fields. That's the "menu" the main agent reads to compose a runtime API call (`exec_code` → `https://api.airtable.com/v0/appXyz/tblAbc?filterByFormula=...`). Same idea for Linear (mirror = team list + project list + issue headers; NOT issue descriptions), GitHub (mirror = repo list + issue/PR headers; NOT file contents), and so on. + +This schema_doc gets included in the structure-skill so the main agent knows the shape without having to infer it. + +### 3.5 The integration skill — `skills/integrations/.md` in R2 + +Markdown file with frontmatter. Every connected integration has one — this is the universal contract between the sync agent (writer) and the main agent (reader). Two flavors, distinguished by the `type` frontmatter: + +- `type: sync-engine` — integration has a real sync engine + mirror. Body covers mirror shape, query patterns, content-fetch patterns, write patterns. +- `type: api-only` — integration is API-only (no sync engine). Body covers the API surface with `exec_code` snippets. + +#### 3.5.1 Sync-engine flavor (Dropbox example) + +```markdown +--- +name: dropbox +description: Folder layout, naming conventions, and query patterns for this user's Dropbox +type: sync-engine +integration: dropbox +sync_id: 01HW...XYZ +mirror_path: integrations/dropbox/struct.json +skill_version: 1 +generated_at: 2026-05-17T18:30:12Z +--- + +# Dropbox structure + +The mirror at `integrations/dropbox/struct.json` is a flat entries array... +[schema_doc embedded here] + +## Folder layout + +Everything lives under two roots: + +### /Personal +- `/Personal/Tax/` — tax filings, named `YYYY-form-name.pdf` +- `/Personal/Visa/` — immigration documents +- `/Personal/Family/` — family important docs (passports, birth certificates...) +- `/Personal/Investments/` — personal angel checks, K-1s +- `/Personal/Leases/` — current and historical lease agreements + +### /Fund +- `/Fund//Portfolio//` — one folder per portfolio company + - Each contains `one-pager.md`, board decks under `Board/`, financials under `Financials/` +- `/Fund//LP/` — LP-facing materials +- `/Fund//Operations/` — internal ops + +## What's in the mirror vs. fetched on demand + +In the mirror (`integrations/dropbox/struct.json`): file/folder paths, names, ids, types, sizes, modified-at, mime-types. **NOT** file contents. + +To get a file's contents, `exec_code` the Dropbox download API (snippet under "Reading content" below). + +## Query patterns (mirror-only — no API call needed) + +- Find tax filings: read `integrations/dropbox/struct.json`, filter `entries` where `path` starts with `/Personal/Tax/` +- Find a portfolio company's docs: filter `path` matches `/Fund//Portfolio//` +- Find the most recent board deck for a company: filter `path` matches `/Fund/*/Portfolio//Board/`, sort by `modified` desc + +## Reading content (mirror gives the path; exec_code fetches the bytes) + +```js +// In exec_code: +const path = "/Personal/Tax/2024-form-1040.pdf"; // resolved from mirror +const r = await fetch("https://content.dropboxapi.com/2/files/download", { + method: "POST", + headers: { + "X-Pd-App": "dropbox", + "Dropbox-API-Arg": JSON.stringify({ path }) + } +}); +const bytes = new Uint8Array(await r.arrayBuffer()); +// process / decode / return as needed +``` + +## Writing files + +To upload, the agent should use `exec_code` with the Dropbox API: + +```js +// In exec_code: +const body = new Uint8Array(...); // file contents +const r = await fetch("https://content.dropboxapi.com/2/files/upload", { + method: "POST", + headers: { + "X-Pd-App": "dropbox", + "Dropbox-API-Arg": JSON.stringify({ path: "/Personal/Tax/2025-form-1099.pdf", mode: "add" }), + "Content-Type": "application/octet-stream" + }, + body +}); +``` + +Webhook will fire and the mirror updates within ~30s. + +## Caveats + +- Old `/Old` folder is deprecated — user said to ignore everything under it. +- `Drafts/` subfolders contain WIP — confirm with user before relying on these as authoritative. +``` + +The skill is written **once** by the sync agent at install time and **updated** if `update_sync` runs (e.g., the user reorganizes and re-triggers via `/sync rerun dropbox`). + +#### 3.5.2 API-only flavor (MailChimp example) + +For integrations the sync agent triaged into Branch B, no mirror exists. The skill is a focused API-usage reference: + +```markdown +--- +name: mailchimp +description: Broadcast email + transactional API for the user's MailChimp account +type: api-only +integration: mailchimp +skill_version: 1 +generated_at: 2026-05-17T18:30:12Z +--- + +# MailChimp + +MailChimp is a broadcast email service. The user's account has 3 audiences (lists) and ~12 active campaigns. There's no structural surface worth mirroring locally — every operation is a focused API call. + +## Common operations + +### List audiences + +```js +// In exec_code: +const r = await fetch("https://us21.api.mailchimp.com/3.0/lists", { + headers: { "X-Pd-App": "mailchimp" } +}); +const data = await r.json(); +// data.lists[*] -> { id, name, contact, stats } +``` + +### Send a campaign + +```js +// In exec_code: +// 1) Create the campaign +const create = await fetch("https://us21.api.mailchimp.com/3.0/campaigns", { + method: "POST", + headers: { "X-Pd-App": "mailchimp", "Content-Type": "application/json" }, + body: JSON.stringify({ + type: "regular", + recipients: { list_id: "" }, + settings: { subject_line: "...", from_name: "...", reply_to: "..." } + }) +}); +const campaign = await create.json(); + +// 2) Set content +await fetch(`https://us21.api.mailchimp.com/3.0/campaigns/${campaign.id}/content`, { + method: "PUT", + headers: { "X-Pd-App": "mailchimp", "Content-Type": "application/json" }, + body: JSON.stringify({ html: "

...

" }) +}); + +// 3) Send +await fetch(`https://us21.api.mailchimp.com/3.0/campaigns/${campaign.id}/actions/send`, { + method: "POST", + headers: { "X-Pd-App": "mailchimp" } +}); +``` + +### Add a subscriber + +```js +const r = await fetch(`https://us21.api.mailchimp.com/3.0/lists//members`, { + method: "POST", + headers: { "X-Pd-App": "mailchimp", "Content-Type": "application/json" }, + body: JSON.stringify({ email_address: "user@example.com", status: "subscribed" }) +}); +``` + +## Docs + +- Official API reference: https://mailchimp.com/developer/marketing/api/ +- All endpoints under `https://.api.mailchimp.com/3.0/` where `` is the user's data center (e.g., `us21`). + +## Caveats + +- Datacenter prefix (`us21`) is fixed per account; embedded in URLs above. +- No webhooks for receive-side events in v1 — pure write-only usage. +``` + +That's the entire deliverable for a Branch B integration. No mirror, no sync_engine row, no facet. ~40 lines. + +### 3.6 The handler runtime contract + +The sync agent writes a single ES module string and stores it in `sync_engine.handler_js`. Worker Loader compiles it on demand; the bundle exports a `WorkerEntrypoint`-shaped class whose methods correspond to the handler's exports. + +**v1 exports (the agent provides any subset that fits the integration):** + +```ts +// Required for all integrations +export async function backfill(ctx: HandlerCtx): Promise; + +// For webhook integrations (Archetypes A and B) +export async function verifySignature(body: string, headers: Record, secret: string): Promise; +export async function handleWebhook(req: Request, ctx: HandlerCtx): Promise; +export async function registerWebhook(ctx: HandlerCtx, args: { callbackUrl: string, label: string }): Promise<{ id: string, secret: string, expiresAt?: number }>; +export async function unregisterWebhook(ctx: HandlerCtx, args: { id: string, label: string }): Promise; + +// For Archetype B (expiring channels) +export async function renewWebhook(ctx: HandlerCtx, args: { id: string, label: string }): Promise<{ id: string, expiresAt: number }>; + +// For Archetype C (polling) +export async function pollOnce(ctx: HandlerCtx): Promise; +``` + +**`HandlerCtx` (supervisor-injected, minimal surface):** + +```ts +interface HandlerCtx { + // Egress: every fetch must set X-Pd-App; the supervisor's HttpGateway RPC + // wraps this so the facet can't issue raw outbound fetches. + fetch: typeof fetch; + + // R2 mirror access — RPC back to the supervisor (which owns the R2 binding). + // Scoped to this sync's mirror_path; the facet can't read or write any + // other sync's mirror. + readMirror(): Promise; + writeMirror(j: unknown): Promise; + + // Facet's own SQLite — direct access. Tables defined by the sync agent's + // schema_ddl, run by the supervisor at facet bootstrap (§3.7). Used for: + // - agent_dedup tables (idempotency keys, sauna pattern) + // - raw_events table during the probe loop (phase 1 capture) + // - cursor state when the agent prefers SQL over a field in the mirror json + // - retry/backoff bookkeeping + // - any per-handler-invocation state the agent decides matters + // NOT the main-agent-facing data layer (that's the r2 mirror). + sql: SqlStorage; + + // Alarm scheduling lives in the supervisor (§3.9) — facets can't call + // setAlarm (empirically validated; throws "internal error" on CF runtime). + // The handler's renewWebhook/pollOnce are invoked by the supervisor's + // alarm() body via RPC. Handler's return values (e.g., new expiresAt + // from renewWebhook) are consumed by the supervisor to recompute the + // next alarm. + + waitUntil(p: Promise): void; + + syncId: string; + integration: string; + accountId: string; + webhookSecret: string | null; + + logger: { info(msg: string, fields?: Record): void; error(msg: string, fields?: Record): void }; +} +``` + +That's it. No supervisor SQL access (only facet's own sqlite). No sub-agent spawning. No process or filesystem access beyond mirror RPCs. The handler is a pure async function with fetch + mirror read/write + scratch sqlite + self-alarm. + +**Why this is enough for Dropbox:** the handler reads the current mirror (gets `_cursor`), calls `https://api.dropboxapi.com/2/files/list_folder/continue` with the cursor, receives a list of file changes, applies them to the entries array, updates `_cursor`, writes the mirror back. ~30 lines of JS. Cursor could also live in `ctx.sql` if the agent prefers — choice is the agent's. + +### 3.7 Per-sync facet (Cloudflare DO Facets primitive) + +Each sync runs inside its own DO facet, created on demand from the Kernel DO supervisor via `this.ctx.facets.get(sync.id, callback)`. Facets are Cloudflare's primitive for "child DO instances managed by a parent, with isolated SQLite per facet, addressable only through the parent." Reference: https://developers.cloudflare.com/dynamic-workers/usage/durable-object-facets/ + +**Wrapper bundle (agent-authored module + supervisor scaffolding):** + +The handler module the agent writes is wrapped by a supervisor-authored facet entrypoint (modeled on sauna's `src/facet/wrapper.ts`). The wrapper exports a class extending `DurableObject` (which is what `getDurableObjectClass("App")` returns from a Worker Loader bundle) that: + +- Has one RPC method per agent-authored export (`verifySignature`, `handleWebhook`, `registerWebhook`, `unregisterWebhook`, `renewWebhook`, `pollOnce`, `backfill`). +- Each RPC method forwards args to the user module's named export, after injecting `HandlerCtx`. +- **Bootstraps the facet's SQLite tables on first invocation** by running `sync_engine.schema_ddl` against `ctx.storage.sql`. Uses `CREATE TABLE IF NOT EXISTS` semantics so it's idempotent across facet wake-ups. Run once per facet lifecycle (the DDL itself is `IF NOT EXISTS`-guarded so re-runs are no-ops). On `update_sync`-bumped version, the new DDL runs against the existing tables — agent must include `CREATE TABLE IF NOT EXISTS` + `ALTER TABLE` migrations as needed (sauna pattern: `migration_ddl` is part of `update_sync`'s contract). +- **Does NOT implement `alarm()`** — facets cannot use the alarm API (empirically validated; see §3.9). Renewal and poll dispatches arrive as ordinary RPC calls from the supervisor when its own alarm fires. +- Catches handler throws, RPCs them back to the supervisor as `sync_errors` rows (with payload preview, error message, stack trace, handler version), re-throws so the parent's `.fetch()` / RPC call sees the error. + +**Capabilities passed at facet creation (sauna pattern):** + +```ts +// Inside Kernel DO supervisor, when dispatching a webhook or scheduling work: +const facet = this.ctx.facets.get(sync.id, async () => { + const codeId = `sync:${sync.id}:v${sync.version}`; + const worker = await this.env.LOADER.get(codeId, async () => buildHandlerBundle(sync.handlerJs)); + const HandlerClass = worker.getDurableObjectClass("Handler"); + return { + class: HandlerClass, + // Capabilities the facet receives — facet code cannot construct these itself. + // sauna pattern: HttpGateway proxies the Pipedream-routed fetch; MirrorIO + // scopes r2 access to this sync's mirror_path only. + httpGateway: this.ctx.exports.HttpGateway({ props: { integration: sync.integration, accountId: sync.accountId } }), + mirrorIO: this.ctx.exports.MirrorIO({ props: { mirrorPath: sync.mirrorPath } }), + // Static metadata baked into wrapper as constants. + syncId: sync.id, + integration: sync.integration, + accountId: sync.accountId, + webhookSecret: sync.webhookSecret, + strategy: sync.strategy, + renewalIntervalSec: sync.renewalIntervalSec, + pollIntervalSec: sync.pollIntervalSec, + }; +}); +``` + +`HttpGateway` and `MirrorIO` are `WorkerEntrypoint` classes exported by the Kernel Worker — the same pattern agent-os already uses for `PipedreamProxy`, `BrowserBridge`, `ExecRunner`, `Fs`. Each is a thin RPC surface that the facet code calls; the supervisor enforces auth + scoping inside the RPC methods. + +**Worker Loader caching:** + +`env.LOADER.get(codeId, async () => buildHandlerBundle(sync.handlerJs))` caches the compiled bundle by `codeId = "sync::v"`. Bumping `version` (via `update_sync`) invalidates the cache cleanly. Warm-start latency for subsequent deliveries: negligible (~10ms reuse). + +**Lifecycle:** + +- **Create:** first `ctx.facets.get(sync.id, ...)` call instantiates the facet and runs the wrapper's constructor. +- **Resume:** subsequent calls after hibernation skip the callback (facet wakes from its existing state). +- **Update:** `update_sync` bumps version + calls `this.ctx.facets.abort(sync.id, "update")`. Next dispatch re-runs the callback with the new code. The facet's SQLite (dedup tables, etc.) is preserved across abort — only running code is killed. +- **Delete:** `delete_sync` calls handler's `unregisterWebhook` (via a final RPC), then `this.ctx.facets.delete(sync.id)`. This shuts down the facet AND drops its SQLite. Then the supervisor deletes the R2 mirror + skill and updates the row to `status='deleted'`. + +**File:** `apps/kernel/src/sync/facet.ts` — exports `getOrCreateFacet(kernel, sync)`, `buildHandlerBundle(handlerJs)`, the wrapper entrypoint factory. + +### 3.8 Webhook receiver route + +Hono route in `apps/kernel/src/index.ts`: + +``` +POST /webhook/:integration/:account_id +``` + +Path strategy: +- `:integration` is the Pipedream app slug (e.g., `dropbox`). +- `:account_id` is the Pipedream account_id (the per-user identifier — same value as the `` block surfaces). + +Why this path: it embeds both the routing key and a quasi-secret (account_id is not high-entropy but is non-obvious). Combined with the handler's `verifySignature` HMAC check, it's enough auth surface for v1. We do NOT trust the path alone — the signature is the auth check. + +Route handler: +1. Look up the Kernel DO singleton by id `"default"`. +2. Call `kernel.dispatchWebhook(integration, accountId, bodyText, headers)`. +3. Return whatever the DO returns (2xx for verified deliveries, 401 for bad signature, 404 for unknown sync, 500 for handler throws). + +`Kernel.dispatchWebhook`: +1. SELECT `sync_engine` WHERE integration AND account_id AND status='active'. If none, return 404. +2. `const facet = getOrCreateFacet(this, row)`. +3. `await facet.verifySignature(bodyText, headers, row.webhookSecret)`. If false, INSERT `sync_errors { error_message: "signature verification rejected", payload_preview: truncated(bodyText) }` and return 401. +4. `await facet.handleWebhook(bodyText, Object.fromEntries(headers))`. (Facet wrapper reconstructs the Request inside; we don't pass Request objects across RPC.) Catch throws → INSERT `sync_errors { error_message, stack_trace, payload_preview }` + 500. +5. Update `sync_engine.lastSyncedAt`. Return 200 (or whatever status the handler returned, surfaced via RPC). + +### 3.9 Central alarm scheduler in the Kernel DO supervisor + +**Why central, not per-facet:** empirically validated 2026-05-17 against Cloudflare's runtime — **facets do NOT support `setAlarm`**. A facet calling `ctx.storage.setAlarm(at)` throws an opaque "internal error". `ctx.storage.put/get` work fine in facets; only the alarm API is unsupported. Cloudflare docs are silent on this. Validation worker: `cf-facet-alarm-test.pandaronit25.workers.dev`. + +So the supervisor (Kernel DO) owns the single alarm. Facets receive renewal/poll dispatches via RPC from the supervisor when its alarm fires. + +**Supervisor's `alarm()` body (in `apps/kernel/src/kernel.ts`):** + +1. Read all `sync_engine` rows with `status='active'` and (`strategy='webhook_channel'` AND `webhook_expires_at IS NOT NULL`) OR (`strategy='poll'`). +2. For each, in sequence: + - Channel renewal: if `webhook_expires_at - now < RENEWAL_BUFFER_MS`, RPC into the facet's `renewWebhook` method. On success, update `webhook_expires_at` and `last_synced_at`. On throw, INSERT `sync_errors` row (payload_preview null), apply backoff (30s → 5min → 30min before next attempt; after 3 in a row, status='errored'). + - Poll dispatch: if `now - last_polled_at >= poll_interval_sec * 1000`, RPC into the facet's `pollOnce` method. On success, update `last_polled_at` and `last_synced_at`. Errors handled identically. +3. Compute the next alarm time: + `nextAlarm = min(min(webhook_expires_at - RENEWAL_BUFFER_MS), min(last_polled_at + poll_interval_sec*1000), now + IDLE_TICK_MS)` + The `IDLE_TICK_MS` ceiling (5 min) guarantees the supervisor re-evaluates periodically even when no syncs are scheduled — picks up newly-created syncs without needing a wake-up. +4. `this.ctx.storage.setAlarm(nextAlarm)`. + +**On `create_sync_engine`:** after the facet is bootstrapped and `backfill` returns, the supervisor calls `recomputeNextAlarm()` which calls `setAlarm` with the new minimum. A `webhook_channel` sync's first renewal gets scheduled immediately; a `poll` sync's first poll gets scheduled `poll_interval_sec` from now. + +**On `update_sync` changing strategy or intervals:** after the facet is re-instantiated with the new code, supervisor calls `recomputeNextAlarm()` again. + +**On `delete_sync`:** supervisor recomputes the alarm to drop the deleted sync's contribution to the min. + +**`RENEWAL_BUFFER_MS = 10 * 60 * 1000`** (10 minutes). Google Drive channels live up to a week; 10-min buffer is way more than enough but tolerant of brief DO downtime. +**`IDLE_TICK_MS = 5 * 60 * 1000`** (5 minutes). + +**Trade-off vs. per-facet alarms:** central scheduler means one alarm storage slot serves N syncs; recomputing the minimum after every state change is O(active syncs), trivial at the scales we'll see (<100). Loses the property of each sync owning its own cadence in isolation, but that wasn't actually achievable on Cloudflare's runtime today. + +### 3.10 The sync agent + +The sync agent is **not** a tool exposed to any LLM. It runs as a queue-consumer-triggered sub-agent inside the Kernel DO. The user never types `/sync ` to launch it — connecting the integration in `/connect` is the trigger. + +**Trigger pipeline:** + +``` +User completes /connect in browser + ↓ +Kernel observes new Pipedream account (via existing diffAccounts + /connect refresh poll OR direct client signal — see §3.10.1) + ↓ +Kernel publishes to `integration-sync` queue: + { type: "integration.connected", integration: "dropbox", account_id: "acc_xyz" } + ↓ +Queue consumer (apps/kernel/src/queue/integration-sync-handler.ts) receives the message + ↓ +Consumer RPCs: env.KERNEL.get(idFromName("default")).runSyncAgent("dropbox", "acc_xyz") + ↓ +Kernel DO: idempotent guard — does skills/integrations/dropbox.md exist in R2? + If yes: no-op (sync agent already ran for this slug; use `/sync rerun` to force). + If no: proceed. + ↓ +Kernel DO: ensure "system" thread exists, create sync agent run row (kind="sync"), call runSubAgent({systemPrompt: SYNC_AGENT_SYSTEM_PROMPT, tools: buildSyncAgentTools(perTurn, {integration, accountId}), ...}) + ↓ +Sync agent: + Phase 1 — Triage. Is this integration sync-able (has structural surface worth mirroring)? + ├── YES → Phases 2–7 (schema, skill, handler, install, probe, verify). Writes skill md + struct.json, creates sync_engine row + facet. + └── NO → Phase 2-API. Writes a short API-only skill md (no sync_engine row, no facet, no mirror). Exits in ~30 seconds. + ↓ +On completion: queue ack. On failure: log + DLQ retry policy applies. +``` + +**Why a queue rather than direct call:** decouples connection-event detection from agent execution. The sync agent run can take minutes (sync-able branch) or 30 seconds (API-only branch) and burns tokens; running it on the connection-detection path would block other connection updates and overflow Worker CPU/wall-time limits. Queue gives us retries, DLQ for poison messages, and back-pressure for free. + +**Why run on every connection rather than gating by integration type:** 3,000+ Pipedream integrations and counting. Maintaining a hand-curated allowlist of "sync-able" slugs would (a) rot fast, (b) miss new integrations, (c) can't reason about *this user's* specific use of a flexible integration like Notion (could be a wiki, could be a database, could be both). The agent is the only piece capable of looking at the actual integration and deciding. The cost of running the agent on a write-only API like MailChimp is ~30 seconds of one cheap turn — small enough to absorb. + +**Why a "system" thread for the run:** sync agent runs need to persist their transcript for debuggability (run row + messages, same as `spawn_sub_agent`). They don't belong in any user chat thread — they weren't typed there. v1 routes them to a special thread with id `"system"` that the kernel creates lazily. v2 may expose this via `/system` command for browsing. For v1 the user doesn't see the transcript directly; only the resulting sync_engine row, skill md, and mirror json are user-facing. + +**Run shape:** same `runSubAgent` helper that `spawn_sub_agent` uses (extracted in PR-F2). Differences: +- `systemPrompt`: `SYNC_AGENT_SYSTEM_PROMPT` (port of sauna's, adapted for json+md output) +- `tools`: `buildSyncAgentTools(perTurn, {integration, accountId})` — curated set per §3.10.3 +- `threadId`: `"system"` (the system thread) +- `runKind`: `"sync"` (new kind alongside `main` / `sub`) +- `model`: `"opus"` (per the memory: opus default on agent-os for subagents; sync work is high-leverage and bug-sensitive) +- No `task_id` linkage (sync agent isn't running on behalf of a thread task) + +### 3.10.1 Connection-event detection + +Two paths converge on the same outcome (queue publish): + +**A. Kernel-side poll (already exists).** `/connect` slash command today triggers `diffAccounts()` against Pipedream — new accounts are detected and persisted. PR-F1 hooks the diff: whenever a new row lands in the integration-connections table, kernel publishes `{type: "integration.connected", ...}` to the queue. + +**B. Client-emitted signal (faster path).** After the user completes the Pipedream OAuth in the browser, the CLI POSTs `/integrations/connected` to the kernel with the integration slug + account_id (the CLI knows these from the Pipedream redirect). Kernel publishes the queue event immediately. This avoids the round-trip delay of waiting for the next `/connect` poll. + +Both paths land at the same queue. The queue consumer is idempotent (it's gated by the existing-sync check inside `runSyncAgent`), so duplicate events from A + B are safe. + +### 3.10.2 Failure + retry semantics + +Queue consumer ack semantics: +- **Successful sync agent completion** (skill md written, regardless of branch) → ack. +- **Sync agent fails before writing the skill** (early discovery failure, model API error, etc.) → ack the message (do NOT retry automatically — sync agent failure is usually deterministic; user retries via `/sync rerun `). +- **Infrastructure failure** (kernel RPC throws, queue handler timeout) → nack → queue retries with exponential backoff. After max retries, DLQ. +- **Idempotency:** consumer checks `skills/integrations/.md` existence in R2 before starting. If skill exists, returns no-op (the agent already ran). The skill IS the trace — present skill means present decision (engine OR api-only), past trip. + +### 3.10.3 Sync agent system prompt (sketch) + +Heavy port of `sauna-assignment/src/agent/system-prompt.ts`, restructured for the json+md output, with a triage gate up front and the index-not-content principle baked into the sync-engine branch. Sections: + +**Identity + mission.** "You are the sync agent. You are invoked when the user connects a new third-party integration via `/connect`. Your job: figure out whether this integration is worth a full sync engine, and write a skill file at `skills/integrations/.md` that teaches the main agent how to use the integration. If a sync engine is worth building, also build it. Always write the skill, even when the engine isn't built." + +**The two branches.** Every run takes exactly one path through the procedure: + +- **Branch A — Sync engine (sync-able integrations).** File systems, databases, ticket trackers, wikis, calendars, messaging systems. The integration has a navigable structure that the user benefits from indexing locally. Phases 1 → 2 → 3 → 4 → 5 → 6 → 7. Outputs: skill md + struct.json + sync_engine row + active facet. +- **Branch B — API-only (everything else).** Write-only APIs (MailChimp, Twilio, Sendgrid), AI APIs (OpenAI, Anthropic), payments (Stripe-write), social broadcasters (LinkedIn, WhatsApp), most read-it's-cheap APIs. Phase 1 → Phase 2-API → exit. Outputs: skill md only. + +#### Phase 1 — Triage (every run starts here) + +Use `fetch_docs` against the integration's API documentation and (sparingly) `pipedream_fetch` to enumerate the integration's top-level resource types. Ask the following questions, in order: + +1. Does the integration EXPOSE structural primitives that the user navigates? (Folders, tables, projects, calendars, channels — durable container-like resources.) If no → **Branch B**. +2. Is the user likely to ASK questions like "where is X?" / "what's in Y?" / "find me Z by predicate" against the integration? If no → **Branch B**. +3. Would knowing the structural layout locally save material work at runtime (5+ API calls, multi-second latency) compared to just calling the API ad hoc? If no → **Branch B**. + +If all three are yes → **Branch A**. Otherwise → **Branch B**. When unsure, choose Branch B — it's the cheap option, and the user can `/sync rerun ` later if they want to escalate. + +Examples to anchor the decision (the prompt embeds these): +- Dropbox → A (folder/file tree, "find tax filings", saves recursive listing) +- Airtable → A (bases/tables/fields, "find Series-A companies", saves schema discovery) +- Linear → A (teams/projects/issue headers, "what's open on the auth refactor?", saves list-issue scans) +- Notion → A if the user has databases/wiki pages worth indexing; B if it's just a private journal — agent should peek to decide +- MailChimp → B (broadcast API, no structural navigation) +- LinkedIn → B (read your own profile + post; no folder-tree) +- WhatsApp → B (send messages) +- OpenAI → B (model-call API; nothing to index) +- Stripe → B for v1 (payment writes; could be A later for customer/subscription listings, but defer) +- Slack → A for v1's purpose, but borderline; prompt encourages Branch A if the user has many channels and asks "what did say last week" types of questions + +#### Phase 2-API — Branch B skill writing (terminal phase for Branch B) + +Write `skills/integrations/.md` with frontmatter `type: api-only` and a body of ~30-80 lines covering: +- One-paragraph identity ("MailChimp is a transactional + broadcast email API.") +- Common operations with `exec_code` snippets — each snippet uses `X-Pd-App: ` and shows the API call. +- Cross-references to the integration's official docs URL. +- Caveats / scope. + +Then exit. No `create_sync_engine` call. No struct.json. The skill alone is the deliverable. + +#### Phase 2 — Branch A: PRIME DIRECTIVE re-read + schema design + +(This and the following phases run only for Branch A.) + +THE PRIME DIRECTIVE — re-read before designing the schema: + +> You are not building a backup. You are building an **index** — the minimum local data that makes the user's integration efficiently navigable. Think like a senior engineer wiring up a sync engine for this specific integration: what's the navigational layer (the menu) vs. the content layer (what you order from the menu)? +> +> - File systems: paths + names + ids + sizes + modified-at + mime. NOT file bytes. +> - Databases: bases + tables + fields + select-options + relationships. NOT records / rows. +> - Ticket trackers: teams + projects + issue headers (id, title, state, labels, assignee, updated_at). NOT issue bodies, NOT comments. +> - Wikis: page tree + titles + parents + last-edited-at. NOT page bodies. +> - Calendars: calendars + recent event headers. NOT descriptions / attachments. +> - Messaging: workspaces + channels + recent message headers. NOT bodies. +> +> Back off if you're tempted to mirror anything expensive-to-re-fetch but trivial-to-fetch-on-demand. The main agent's job is to **read your mirror to figure out WHAT to ask for, then `exec_code` the upstream API to actually fetch it.** Your mirror enables that step-1 lookup. +> +> Sanity check: imagine the user has 50,000 records / 100GB of files / 10,000 issues. Does your mirror stay under ~5MB? If not, you're mirroring content. + +Decide the mirror json shape based on what discovery surfaced, applying the Prime Directive ruthlessly. Write the `schema_doc` prose. The schema_doc must explicitly call out "what's in the mirror vs. fetched on demand" so the main agent knows both halves. + +Then decide the **facet SQLite schema** — separate from the R2 mirror. Write `schema_ddl` as a SQL string with `CREATE TABLE IF NOT EXISTS` statements. This is the handler-internal state layer; not visible to the main agent. Typical contents: + +```sql +-- Mandatory if your handler does anything idempotency-sensitive (sauna pattern): +CREATE TABLE IF NOT EXISTS agent_dedup ( + dedup_key TEXT PRIMARY KEY, + claimed_at INTEGER NOT NULL +); + +-- Optional: cursor in SQL if you'd rather not embed it in the mirror json. +-- For Dropbox: leave cursor in the mirror's _cursor field — simpler. +-- For more complex multi-cursor integrations (one cursor per Google Drive change-channel), +-- a table is cleaner: +CREATE TABLE IF NOT EXISTS cursors ( + resource_id TEXT PRIMARY KEY, + cursor TEXT NOT NULL, + updated_at INTEGER NOT NULL +); + +-- For the probe-loop's phase-1 raw capture (Phase 6): +CREATE TABLE IF NOT EXISTS raw_events ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + received_at INTEGER NOT NULL, + action TEXT, + type TEXT, + payload_json TEXT NOT NULL +); +``` + +Empty string is allowed for handlers that genuinely need no SQL state. The supervisor runs the DDL once at facet bootstrap; the handler can then `INSERT/SELECT/UPDATE` via `ctx.sql`. + +#### Phase 3 — Skill authoring (Branch A) + +Write `skills/integrations/.md` with frontmatter `type: sync-engine` and a body covering: +- Readme — how the user organizes things (folder layout, table purposes, etc.) +- Schema doc — what's in the mirror vs. what's fetched on demand +- Query patterns — `read_file` + grep snippets for common lookups against the mirror +- Reading content on demand — `exec_code` snippets given an id from the mirror +- Writing — `exec_code` snippets for create/update/delete upstream +- Caveats + +#### Phase 4 — Handler authoring (Branch A) + +Pick the strategy (`webhook_stable` / `webhook_channel` / `poll`) based on integration capabilities — research with `fetch_docs` if unsure. Write the handler module string. The handler's job is to keep the STRUCTURAL index current — on every webhook delivery it updates the index entries (file moved? issue retitled? record schema changed?), it does NOT pull down content. Required exports for each strategy as in §3.6. + +#### Phase 5 — Install (Branch A) + +Call `create_sync_engine({integration, accountId, strategy, handlerJs, schemaDoc, skillPath, mirrorPath, pollIntervalSec?, renewalIntervalSec?})`. The kernel: compiles via Worker Loader (rolls back on compile failure), calls `handler.registerWebhook` if applicable, calls `handler.backfill` to populate the structural index, sets the initial alarm if needed. + +#### Phase 6 — Probe loop (Branch A; only for unfamiliar webhook shapes) + +If you didn't know the webhook payload shape going in: install a phase-1 raw-capture handler whose schema_ddl includes a `raw_events` table (see Phase 2 example) and whose `handleWebhook` does `INSERT INTO raw_events (received_at, action, type, payload_json) VALUES (...)` via `ctx.sql.exec(...)`. Mirror writes are skipped in this phase. Trigger one real event (or `run_in_sync` an API call that fires one). Use `query_sync(sync_id, "SELECT * FROM raw_events ORDER BY received_at DESC LIMIT 10")` to read the captured payloads, parse them, design the real schema + handler + mirror shape, then `update_sync` with the production handler + new `schema_ddl` + a `migration_ddl` that drops `raw_events` and creates whatever the production handler needs. + +#### Phase 7 — Verify (Branch A) + +`read_mirror` and confirm shape matches schema_doc AND adheres to the Prime Directive. `get_sync_errors` to confirm no errors. Done. + +The prompt cribs the sauna prompt's behavioral guidance heavily — payload-vs-contracts split (probe loop for payload shapes, fetch_docs for signing algorithms), the "don't guess field names from training data" warning, response style. + +### 3.10.4 Sync agent tool surface + +Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`: + +**Read tools:** +- `pipedream_fetch({host, path, method, headers, body, app_slug})` — a thin wrapper over a `fetch` that sets `X-Pd-App: ` and goes through Pipedream proxy. Returns `{status, headers, body_text}`. Same proxy `exec_code` uses. +- `web_search(query)` — existing. +- `fetch_docs(url)` — existing (Firecrawl-backed). +- `read_mirror(sync_id)` — reads `integrations//struct.json` via R2. +- `query_sync(sync_id, sql)` — SELECT-only SQL against the facet's own SQLite. Used to read `raw_events` during the probe loop, inspect `agent_dedup` state during debugging, etc. Routed via supervisor RPC into the facet's SQL surface. +- `get_sync_errors(sync_id, limit?)` — paginated rows from `sync_errors`, newest first, by `(sync_id, occurred_at DESC)`. Returns `{occurred_at, error_message, stack_trace, payload_preview, handler_version}` per row. Used by the sync agent during the probe loop (verify phase-2 signature checks) and by `update_sync` debugging. + +**Write tools (all of these mutate the kernel DO or R2):** +- `write_skill({slug, body})` — writes `skills/integrations/.md`. Validates frontmatter. +- `create_sync_engine({integration, account_id, strategy, handler_js, schema_ddl, schema_doc, skill_path, mirror_path, poll_interval_sec?, renewal_interval_sec?})` — inserts the row in status='discovering', compiles the bundle via Worker Loader (rolls back on compile error), boots the facet (which runs `schema_ddl` against the facet's SQLite), calls `handler.registerWebhook` if applicable, calls `handler.backfill`, sets first alarm. On success flips status to 'active'. On any failure: status='errored', errorText populated, returns the error. +- `update_sync({sync_id, handler_js?, schema_ddl?, migration_ddl?, schema_doc?, strategy?, poll_interval_sec?, renewal_interval_sec?})` — atomic update. Bumps `version`. If `handler_js` changes, recompiles. If `schema_ddl` changes, runs `migration_ddl` against the existing facet SQLite BEFORE swapping (sauna pattern; agent provides the migration as `ALTER TABLE` / data-move statements). On migration failure, rolls back the whole update. If strategy changes from `poll` → `webhook_*`, registers webhook; vice versa unregisters. +- `run_in_sync({sync_id, code})` — one-shot execution of agent-authored JS inside the handler runtime. Same ctx as handlers. Used for backfill, diagnostics, triggering upstream actions during the probe loop. Output (return value + console logs) returned to the agent. +- `delete_sync({sync_id})` — destructive. Calls `handler.unregisterWebhook` if applicable, deletes `skills/integrations/.md` from R2, deletes `integrations//struct.json`, calls `ctx.facets.delete(sync.id)` (which drops the facet's SQLite — `raw_events`, `agent_dedup`, everything goes), flips sync_engine row to status='deleted' (kept for audit). For Branch-B (api-only) syncs there's no sync_engine row to flip — `delete_sync` just removes the skill. +- `manage_tasks` — existing. The sync agent uses it to track its own phases (good UX: "Phase 1: discovery 🔄 Phase 2: schema 🔄 ..."). + +Tools the sync agent does **not** get: +- `exec_code` — the sync agent's JS authoring happens through `create_sync_engine` and `run_in_sync`; there's no reason to expose raw exec_code (smaller blast radius, clearer audit trail). +- `spawn_sub_agent` — no recursion. +- `computer_use`, `sessions tools` — not relevant. +- fs/git tools — sync agent writes ONLY through the curated `write_skill` and the sync-engine machinery; not free-form file writes. + +### 3.11 Slash commands + +The sync agent is **not** triggered by a slash command. Connecting an integration via `/connect` is the trigger. Slash commands cover only introspection and manual intervention: + +- `/sync status` — lists every integration that has a `skills/integrations/.md`. For each: branch (sync-engine | api-only), and for sync-engine integrations also status (active / errored), strategy, last_synced_at, error_text if any. No LLM call. +- `/sync rerun ` — re-publishes the connection event to the `integration-sync` queue, bypassing the idempotency check (forces re-run even if `sync_engine` row exists in status='active'). Used when (a) a previous run errored, (b) user reorganized their data and wants a fresh discovery, (c) handler bug → resync needed. +- `/sync delete ` — calls the `delete_sync` write tool directly (no agent involvement); confirms with the user first. Unregisters webhook upstream, drops mirror json, drops skill, calls `ctx.facets.delete(sync.id)`, flips row to status='deleted'. + +Implementation: same pattern as existing slash commands (under `apps/cli/src/commands/sync.ts` for the CLI side; server-side dispatch in the kernel). + +### 3.12 api_hosts / egress + +**No new egress gate.** Both the handler runtime's `fetch` and the main agent's `exec_code` `fetch` go through the existing Pipedream proxy. The proxy itself enforces the slug-to-host pairing (a request with `X-Pd-App: dropbox` targeting `api.github.com` is 403'd by Pipedream). The `api_hosts` column from sauna's design is **deliberately omitted**. + +--- + +## 4. Data flows + +### 4.1 Integration-connection → sync agent (end-to-end) + +``` +User completes /connect dropbox in browser + ↓ +CLI or kernel poll detects new account + ↓ +CLI → Kernel API: POST /integrations/connected { integration: "dropbox", account_id: "acc_xyz" } + (or kernel-side diffAccounts detects it directly — either path) + ↓ +Kernel: publish to env.INTEGRATION_SYNC queue: { type: "integration.connected", integration: "dropbox", account_id: "acc_xyz" } + ↓ +Queue Consumer: receives message → env.KERNEL.get(idFromName("default")).runSyncAgent("dropbox", "acc_xyz") + ↓ +Kernel DO: idempotency check on sync_engine for (dropbox, acc_xyz) → none exists, proceed +Kernel DO: ensure "system" thread exists; INSERT run row (kind="sync") in system thread +Kernel DO: runSubAgent({systemPrompt: SYNC_AGENT_SYSTEM_PROMPT, tools: buildSyncAgentTools(perTurn, {integration:"dropbox", accountId:"acc_xyz"}), prompt: "", model: "opus", threadId: "system", runKind: "sync"}) + ↓ + [sync agent run begins, transcripts persist in system thread for debugging] + ↓ +SyncAgent: manage_tasks create "Sync dropbox" +SyncAgent: fetch_docs https://www.dropbox.com/developers/reference/webhooks +SyncAgent: pipedream_fetch dropbox /2/users/get_current_account → {account_id, name, country, ...} +SyncAgent: pipedream_fetch dropbox /2/files/list_folder { path: "" } → top-level entries +SyncAgent: (recurse breadth-first, build mental model) +SyncAgent: manage_tasks update "Sync dropbox: Schema" +SyncAgent: (decide on flat-entries shape; write schema_doc) +SyncAgent: manage_tasks update "Sync dropbox: Skill" +SyncAgent: write_skill { slug: "dropbox", body: "" } +SyncAgent: manage_tasks update "Sync dropbox: Handler" +SyncAgent: (write handler_js: verifySignature, handleWebhook (calls list_folder/continue with cursor), registerWebhook, unregisterWebhook, backfill) +SyncAgent: create_sync_engine { + integration: "dropbox", + account_id: "acc_xyz", + strategy: "webhook_channel", + handler_js: "", + schema_doc: "", + skill_path: "skills/integrations/dropbox.md", + mirror_path: "integrations/dropbox/struct.json" + } +Kernel: INSERT sync_engine row, status='discovering' +Kernel: getOrCreateFacet → ctx.facets.get(sync.id, () => { ...buildHandlerBundle → getDurableObjectClass("Handler") }). Wrapper constructor runs. + Wrapper: ctx.storage.sql.exec(sync.schema_ddl) // runs CREATE TABLE IF NOT EXISTS ... for facet's own SQLite + (On Worker Loader compile failure OR DDL syntax error: rollback row, errorText, ctx.facets.delete to clean up partial facet, return error.) +Kernel: recomputeNextAlarm() // supervisor recalculates the min next-alarm across all active syncs and calls this.ctx.storage.setAlarm(at) on itself +Kernel: await facet.registerWebhook({callbackUrl: "https:///webhook/dropbox/acc_xyz", label: "agent-os dropbox sync"}) + → returns {id: "dbid_123", secret: "...", expiresAt: null} +Kernel: UPDATE sync_engine SET webhook_id, webhook_secret, webhook_expires_at +Kernel: await facet.backfill() // facet runs backfill; writes mirror via MirrorIO RPC +Kernel: UPDATE sync_engine SET status='active', last_synced_at + (Facet wrapper's constructor already called setAlarm for renewal/poll if applicable. Dropbox: no alarm.) + ↓ +SyncAgent: manage_tasks update "Sync dropbox: Verify" +SyncAgent: read_mirror sync_id → confirms shape +SyncAgent: get_sync_errors sync_id → empty +SyncAgent: manage_tasks complete "Sync dropbox: Done" + ↓ + [final answer persists as the last message of the sync run in the system thread; queue consumer acks the message] + ↓ + [next time the user types in any thread, dropbox is in , the agent reads skills/integrations/dropbox.md when relevant] +``` + +### 4.2 Webhook delivery (Dropbox, empty ping + cursor fetch) + +``` +Dropbox → POST /webhook/dropbox/acc_xyz (signature in X-Dropbox-Signature, empty body or list_folder accounts) +Kernel route → kernel.dispatchWebhook("dropbox", "acc_xyz", body, headers) +Kernel: SELECT sync_engine WHERE integration='dropbox' AND account_id='acc_xyz' AND status='active' +Kernel: const facet = getOrCreateFacet(this, row) // ctx.facets.get +Kernel: await facet.verifySignature(body, headers, row.webhook_secret) → true +Kernel: await facet.handleWebhook(body, headers) + ↓ (inside facet wrapper → user handler) + const mirror = await ctx.readMirror() // MirrorIO RPC to supervisor + const r = await ctx.fetch("https://api.dropboxapi.com/2/files/list_folder/continue", { + method: "POST", + headers: { "X-Pd-App": "dropbox", "Content-Type": "application/json" }, + body: JSON.stringify({ cursor: mirror._cursor }) + }) // HttpGateway RPC to supervisor → Pipedream → Dropbox + const data = await r.json() + // apply data.entries (added/modified) and data.entries with .tag='deleted' to mirror.entries + // update mirror._cursor = data.cursor + await ctx.writeMirror(mirror) // MirrorIO RPC to supervisor + return new Response("ok", { status: 200 }) + ↓ +Kernel: UPDATE sync_engine SET last_synced_at +Kernel: return 200 to Dropbox // success has no row — sync_engine.last_synced_at advancing IS the success signal +``` + +### 4.3 Main agent query — "where are my tax filings?" (lookup is enough) + +This is the case where the **structural index alone** answers the question — no content fetch needed. + +``` +Main agent's turn begins. +Context block: includes dropbox. System prompt teaches "check skills/integrations/.md". +Skill auto-load: skills/integrations/dropbox.md is in scope. +Main agent decides: this is a dropbox structure question, mirror has the answer. +Main agent: read_file integrations/dropbox/struct.json (returns the full json — file tree, no contents) +Main agent: (inline grep for path starts with /Personal/Tax/) +Main agent: Replies with the list of files + brief summary. +``` + +### 4.3.1 Main agent query that requires content — "summarize last year's tax filings" + +This is the two-step pattern: mirror gives the menu, `exec_code` fetches the content. + +``` +Main agent's turn begins. +Skill auto-load: skills/integrations/dropbox.md is in scope. Skill says: + "To read a file's contents, exec_code: + fetch('https://content.dropboxapi.com/2/files/download', { + headers: { 'X-Pd-App': 'dropbox', + 'Dropbox-API-Arg': JSON.stringify({path: '/path/to/file'}) } + })" +Main agent: read_file integrations/dropbox/struct.json +Main agent: (inline grep for /Personal/Tax/ AND modified > 2025-01-01) + → 4 matching entries, with paths +Main agent: (decides to summarize — needs contents) +Main agent: exec_code with JS that: + - fetches each of the 4 files in parallel via Dropbox download API + - decodes pdf → text via a small library or inline + - returns concatenated text + console.log shows the text content +Main agent: summarizes the text, replies to user. + +Total: 1 mirror read (cheap) + 4 dropbox API calls (focused, only the files we need). +Without the mirror: agent would need to recursively list folders, scan thousands of unrelated files, hope to find the right ones. Far more wasteful. +``` + +This pattern is the actual payoff: the structural index turns "search through everything" into "look up by predicate, then fetch exactly what's needed." + +### 4.4 Main agent write — "create a one-pager for AcmeCorp under the right portfolio folder" + +``` +Main agent reads skill → resolves path: /Fund/Dimension-I/Portfolio/AcmeCorp/one-pager.md +Main agent writes draft markdown in current turn. +Main agent calls exec_code with JS: + const body = new TextEncoder().encode("# AcmeCorp\n...") + const r = await fetch("https://content.dropboxapi.com/2/files/upload", { + method: "POST", + headers: { + "X-Pd-App": "dropbox", + "Dropbox-API-Arg": JSON.stringify({ path: "/Fund/Dimension-I/Portfolio/AcmeCorp/one-pager.md", mode: "add" }), + "Content-Type": "application/octet-stream" + }, + body + }) + console.log(r.status, await r.text()) +exec_code returns success. +~10s later, Dropbox webhook fires → handler updates mirror → next turn the mirror shows the new file. +``` + +--- + +## 5. Failure modes + recovery + +### 5.1 At sync agent runtime +- **Discovery times out / hits cap.** Sync agent writes what it has, marks the sync engine with `status='errored'`, errorText explains. User can `/sync rerun dropbox` to retry (forces re-publish of the connection event, bypassing idempotency). +- **Handler compile failure (Worker Loader rejects the bundle).** Caught inside `create_sync_engine` BEFORE facet construction completes — supervisor calls `ctx.facets.delete(sync.id)` to ensure no partial facet exists, returns the compile error to the agent inline. The agent fixes the handler and re-calls. +- **registerWebhook fails (upstream rejects).** `facet.registerWebhook` throws → supervisor catches → marks status='errored' with the error text → cleans up via `ctx.facets.delete(sync.id)` → returns error to agent. Agent diagnoses (wrong scope, wrong app slug, malformed body), edits handler, re-calls `create_sync_engine`. +- **Backfill throws.** Same shape — but facet stays alive (webhook is already registered upstream). Agent uses `run_in_sync` or `update_sync` to fix and retry. Supervisor doesn't auto-delete on backfill failure because the upstream-registered webhook would need to be unregistered first, which is expensive. + +### 5.2 At webhook delivery +- **verifySignature returns false.** INSERT `sync_errors{error_message:"signature verification rejected", payload_preview}`, return 401. NO sync state change. This is normal scanner noise on a public webhook URL — only worrying if it persists alongside zero successful deliveries (see sauna's debugging rubric, ported into the sync-agent prompt). The rubric: compare `count(sync_errors WHERE occurred_at > T)` against `sync_engine.last_synced_at` — if errors are high AND last_synced_at is stale, the verifier is broken; if errors are high BUT last_synced_at is fresh, it's just scanner noise. +- **handleWebhook throws.** INSERT `sync_errors{error_message, stack_trace, payload_preview}`, return 500. After N consecutive failures (configurable, e.g., 5), flip `sync_engine.status='errored'`. The error is visible to the user via `/sync status` and to the main agent via the sync's skill md (which the agent can re-read to detect the broken-state note appended on errors). User can re-run the sync agent via `/sync rerun `. +- **R2 write fails.** Retry once (transient). Persist failure → 500. + +### 5.3 At facet alarm dispatch +- **renewWebhook throws.** Supervisor's `alarm()` body catches the RPC throw, INSERTs `sync_errors{error_message, stack_trace, payload_preview: null}`, applies backoff (30s → 5min → 30min before next attempt by setting the next alarm accordingly). After 3 consecutive failures, supervisor flips `sync_engine.status='errored'`. +- **pollOnce throws.** Same backoff + flip pattern. +- **Supervisor alarm misses its window** (e.g., Cloudflare incident, deployment churn). Channel expires before renewal fires. On next webhook-or-dispatch attempt, supervisor sees `webhook_expires_at < now` and the handler will throw on the next renewal (or the upstream will 410). Supervisor marks status='errored' with a clear message; user runs `/sync rerun ` to re-bootstrap. + +### 5.4 At Pipedream auth +- **OAuth token revoked / expired.** Pipedream returns 401 to the fetch. Handler propagates as `handler_threw`. After consecutive failures, status='errored', errorText='auth — token may be revoked'. User reconnects via `/connect` and re-runs `/sync`. + +--- + +## 6. Stacked PR slicing (recap) + +### PR-F1 — Foundation primitives (no agent, no queue consumer yet) +**Files created:** +- `packages/models/src/schema/sync-engine.ts` — `sync_engine` + `sync_errors` drizzle tables (full schema as in §3.3) +- `apps/kernel/src/sync/facet.ts` — `getOrCreateFacet(kernel, sync)`, `buildHandlerBundle(handlerJs)`, the facet wrapper entrypoint factory (`Handler` class with one RPC per export + `alarm()` dispatcher) +- `apps/kernel/src/sync/http-gateway.ts` — `HttpGateway` WorkerEntrypoint (Pipedream-proxied fetch, scoped by `props.integration` + `props.accountId`) +- `apps/kernel/src/sync/mirror-io.ts` — `MirrorIO` WorkerEntrypoint (R2 read/write scoped by `props.mirrorPath`) +- `apps/kernel/src/sync/dispatch.ts` — `dispatchWebhook` body (looks up sync row, calls `getOrCreateFacet`, RPCs verifySignature + handleWebhook) +- `apps/cli/src/commands/sync.ts` — `/sync status` and `/sync delete` (no `/sync ` trigger; queue is the trigger) +- `__tests__/fixtures/dropbox-handler.js` — hand-written smoke fixture (verifySignature, handleWebhook with cursor + list_folder/continue, registerWebhook, unregisterWebhook, backfill) + +**Files modified:** +- `apps/kernel/src/index.ts` — add `POST /webhook/:integration/:account_id` route; export `HttpGateway` and `MirrorIO` WorkerEntrypoints alongside existing PipedreamProxy etc. +- `apps/kernel/src/kernel.ts` — add `dispatchWebhook` RPC method, add supervisor-side helper for `sync_errors` row inserts (called via RPC by the facet wrapper on failure) +- `apps/kernel/src/agent/system-prompt.ts` — add the convention sentence: "For any integration in ``, `skills/integrations/.md` (if present) contains usage guidance — read it before invoking that integration." No new injected block. +- `wrangler.jsonc` — declare `HttpGateway` and `MirrorIO` as named WorkerEntrypoints if not auto-discovered + +**Smoke:** install fixture handler via a dev-only RPC that simulates `create_sync_engine` → fixture's `registerWebhook` runs against real Dropbox (Pipedream-proxied) → change a file in Dropbox → empty ping arrives → facet handles → `integrations/dropbox/struct.json` updates within 30s. No agent and no queue consumer involved. + +### PR-F2 — Queue trigger + sync agent + Dropbox e2e +**Files created:** +- `apps/kernel/src/sync/sync-agent-system-prompt.ts` — ported from sauna +- `apps/kernel/src/sync/sync-agent-tools.ts` — `buildSyncAgentTools(perTurn, {integration, accountId})` returning the 9 curated tools +- `apps/kernel/src/queue/integration-sync-handler.ts` — Cloudflare Queue consumer; receives `integration.connected` messages, RPCs `kernel.runSyncAgent` +- `apps/kernel/src/api/integrations-connected.ts` — Hono route `POST /integrations/connected` (CLI-emitted signal path); validates body, publishes to queue +- `apps/kernel/src/sync/run-sub-agent.ts` — extracted helper (factored out of `apps/kernel/src/tools/spawn-sub-agent.ts`'s execute body) + +**Files modified:** +- `apps/kernel/src/tools/spawn-sub-agent.ts` — refactor to use the extracted `runSubAgent` +- `apps/kernel/src/kernel.ts` — add `runSyncAgent(integration, accountId)` RPC method; ensure "system" thread row exists lazily +- `apps/kernel/src/integrations/sync.ts` — extend `diffAccounts()` callsite to publish `integration.connected` events to the queue (kernel-poll path; complement to the CLI-emitted path) +- `apps/cli/src/commands/sync.ts` — add `/sync rerun ` — POSTs to `/integrations/connected` with `force: true` bit that bypasses the idempotency check inside `runSyncAgent` +- `wrangler.jsonc` — declare `integration-sync` queue (producer + consumer bindings; DLQ; max retries) + +**Smoke:** complete `/connect dropbox` in CLI (or simulate by direct POST to `/integrations/connected`) → queue receives event → consumer fires → sync agent run begins in system thread → triage decides Branch A → ~5 min later, `skills/integrations/dropbox.md` (sync-engine flavor) and `integrations/dropbox/struct.json` exist; main agent (in any thread) sees `dropbox` in `` and reads the skill; ask "where are my tax filings?" → answer correct; main agent uses `exec_code` to upload a file → webhook fires → mirror reflects within 30s; `/sync delete dropbox` cleans up upstream + R2. Then ALSO smoke Branch B: connect MailChimp → queue → triage decides Branch B → ~30s later, `skills/integrations/mailchimp.md` (api-only flavor) exists, no sync_engine row, no struct.json. + +### PR-F3 — Polling fallback + research wave + additional integrations +**Files created:** +- (possibly) `apps/kernel/src/sync/pipedream-webhook-bus.ts` — if Pipedream offers a per-user demux solution, this is its client +- additional integration smoke targets in `__tests__/fixtures/` + +**Files modified:** +- `apps/kernel/src/sync/sync-agent-system-prompt.ts` — extend with `pollOnce` + poll-strategy rules +- `apps/kernel/src/sync/sync-agent-tools.ts` — minor extensions if needed + +**Smoke targets:** (each triggered by connecting the integration via `/connect`, then queue consumer dispatches) +- Notion — poll-only (or webhook-bus if research pans out) +- Linear — confirms archetype A works end-to-end with the sauna-style stable webhook +- Google Calendar — confirms expiring-channel renewal fires correctly + +--- + +## 7. Open research items + +1. **Pipedream webhook-bus for Notion-class.** Does Pipedream offer a per-user webhook demux that we can subscribe to and route by account_id? If yes, archetype C collapses into archetype A semantically and we don't need pure polling. +2. **Dropbox webhook channel TTL behavior.** Per user input, Dropbox is "google-drive-like" — but the actual app-config webhook URL doesn't expire. The expiry/renewal question only matters for actual channel-based subscriptions. PR-F1's smoke and PR-F2's e2e will validate Dropbox specifically. +3. **DO storage size of mirrors.** A 10K-entry Dropbox is ~5MB json. A 100K-entry Drive is ~50MB. Where does R2 cost vs. read latency become painful? If it does, sharding strategy (per-top-level-folder json) is a follow-up. +4. **Cost guardrails on sync agent runs.** A bad-luck connection of a 100K-file Dropbox tree could burn $10+ in tokens during discovery. Need a cost ceiling (e.g., "fail loudly if discovery exceeds 100K input tokens") — out of scope for v1 but worth flagging. +5. **Re-explore on drift.** If the user reorganizes their dropbox, the existing skill goes stale. v1 path: user runs `/sync rerun dropbox` (which forces a fresh sync agent run, replacing the skill + handler). v2 could auto-detect drift (skill rewrite triggered when `_cursor` advances past a threshold of folder-structure changes). + +--- + +## 8. Out of scope (v1, deferred to future PRs) + +- ~~Auto-trigger sync agent on `/connect` completion.~~ **In scope as of this spec — that's the entire trigger model (§3.10).** The "out of scope" item now flips to: hardening the cost ceiling on the auto-triggered run so a runaway discovery on a giant integration can't burn unbounded tokens. v1 ships with a soft cap (50 API calls / 5 min target) in the prompt; a hard cap enforced by the runtime lands in F3 or later. +- Schema migrations on `update_sync`. v1 path is delete-and-recreate. v2 could diff the schema_doc and emit a migration plan. +- Multi-account-per-integration. v1 assumes one account per integration slug (one `dropbox` connection per deployment). The schema's unique index supports more, but UX doesn't yet. +- Sub-agents inside handlers (`ctx.agent`). Explicitly out per user input. +- Streaming UI surface for sync-agent runs. Task panel shows coarse status only. +- Cross-integration JOIN tooling. Each mirror stands alone. + +--- + +## 9. Smoke checklists per PR + +### PR-F1 smoke +1. `pnpm dev` → CLI boots, no regressions in existing flows. +2. Drizzle migration applies cleanly. +3. `/sync status` from clean state prints "no active syncs." +4. Dev-only RPC installs the fixture dropbox handler into `sync_engine` with status='active', strategy='webhook_channel', mirror_path='integrations/dropbox-fixture/struct.json'. +5. Fixture's `registerWebhook` runs successfully against real Dropbox API (Pipedream-proxied, against a test account). +6. Webhook URL `https:///webhook/dropbox/` returns 200 to a signed Dropbox POST. +7. Touch a file in Dropbox → empty ping arrives → handler reads cursor, calls list_folder/continue, applies delta → `integrations/dropbox-fixture/struct.json` reflects the change within 30s. +8. Bad-signature request returns 401, `sync_errors` row recorded with `error_message="signature verification rejected"`. +9. Main agent's system prompt has the convention sentence pointing at `skills/integrations/.md`; with the fixture installed, `/sync status` confirms the fixture row exists. +10. `/sync delete dropbox` removes the row, drops the mirror json, fixture's unregisterWebhook is called. + +### PR-F2 smoke +1. From zero state (no fixture, no sync_engine row), simulate the connection event: `curl -X POST http://localhost:8787/integrations/connected -d '{"integration":"dropbox","account_id":"acc_xyz"}'`. +2. Queue receives the message; consumer fires; sync agent run begins (visible as a new run row in the `system` thread, queryable via DB). +3. Sync agent completes within ~5–10 min (target; cost ceiling enforced). +4. `skills/integrations/dropbox.md` exists in R2 and accurately describes the test account's folder structure in prose. +5. `integrations/dropbox/struct.json` exists, has the expected shape, includes top-level folders. +6. Main agent (next turn, in any user thread) sees `dropbox` in `` and reads `skills/integrations/dropbox.md` when relevant. +7. Ask main agent: "where are my tax filings?" → it reads the skill, greps the mirror, answers correctly with paths. +8. Ask main agent: "upload a one-pager for AcmeCorp under the right portfolio folder" → it picks the path from the skill, runs `exec_code`, file lands in Dropbox. +9. Within 30s of step 8, webhook fires, mirror reflects the new file. +10. `/sync status` shows the sync as active with recent last_synced_at. +11. Idempotency: re-POST the same `/integrations/connected` payload → consumer detects existing sync, no-ops. +12. `/sync rerun dropbox` → forces a fresh sync agent run; previous sync deleted first; new run produces fresh skill + mirror. +13. `/sync delete dropbox` unregisters upstream (verified via Dropbox account settings), drops skill + mirror, sync_engine row goes to status='deleted', facet's SQLite dropped via `ctx.facets.delete`. + +### PR-F3 smoke (depends on research outcome) +1. Connecting Notion via `/connect` triggers the sync agent (same queue path), which works end-to-end via whichever strategy research selects (webhook-bus or polling). Main agent can answer Notion-content questions afterwards. +2. Connecting Google Calendar triggers the sync agent, which produces a `strategy='webhook_channel'` sync with renewal. Verify renewal alarm fires before expiry (live or mocked). +3. Connecting Linear triggers a `strategy='webhook_stable'` sync — confirms archetype A works end-to-end without renewal alarms. + +--- + +## 10. Decisions table + +| # | Decision | Rationale | +|---|----------|-----------| +| 0 | **Mirror = structural index, not content** (Prime Directive, §1.1) | The mirror exists to make navigation/lookup cheap. Content (file bytes, page bodies, records, issue descriptions) is fetched on demand via `exec_code`. Mirror sizes stay bounded (target <5MB even on huge integrations), DO/R2 isn't a backup tier, freshness stays a non-issue for content. This is the load-bearing design choice. | +| 1 | Mirror = R2 json blob (not SQLite) | Main agent uses existing `read_file` + grep; no new query primitive needed; concurrency handled by serializing through Kernel DO. | +| 2 | **Per-sync Cloudflare DO Facets** (sauna parity) | Facets give code isolation (handler can't read supervisor's sqlite or env bindings), per-facet alarms for self-scheduled renewal/poll, and built-in lifecycle primitives (abort/delete) that map cleanly to update_sync/delete_sync. No separate billing — facets share the parent DO's billing. Same Worker Loader machinery agent-os already uses for exec_code. | +| 3 | No `api_hosts` allowlist | Pipedream proxy is the egress gate; redundant in agent-os. | +| 4 | No `ctx.agent` in handler runtime | Per user; inline JS handlers are enough; runtime LLM reasoning out of scope. | +| 5 | Agent authors json shape | Sauna parity; "machine builds machine"; schema_doc carries the human-readable description. | +| 6 | Cursor lives inside the mirror json (reserved `_cursor` field) by default; agent can override into facet SQLite via `cursors` table if multi-cursor | Default keeps cursor atomic with mirror writes. For multi-resource integrations (Drive watch channels per folder, etc.), agent can declare a `cursors` table in `schema_ddl` and store it in facet SQLite instead. | +| 6b | Two schemas authored per Branch-A sync: `schema_doc` (prose, describes R2 mirror json) + `schema_ddl` (SQL, defines facet SQLite tables) | Two distinct data layers serve different purposes. Mirror is structural index for main agent (R2 json; grep-friendly). Facet SQLite is handler-internal state (dedup keys, raw_events for probe loop, multi-cursor state). The agent designs both at create_sync_engine time; supervisor runs DDL at facet bootstrap. | +| 7 | Sync agent is NOT a tool — it runs in a Cloudflare Queue consumer triggered by connection event | Decouples connection detection from agent execution. Sync agent runs are minutes-long and burn tokens — running them inline on the connection-detection path would block other work and blow Worker timeouts. Queue gives retries, DLQ, back-pressure for free. | +| 7b | Sync agent runs on EVERY connection; triages first | 3000+ Pipedream integrations; most are API-wrappers with nothing to sync. Hand-curated allowlist would rot fast. Agent's first phase decides "Branch A (full sync engine)" vs "Branch B (skill-only)". Branch B costs ~30s of one cheap run; small enough to absorb. | +| 7c | NO `` injected block; rely on existing `` + skill convention | Pre-injecting metadata for N connected integrations is context bloat at scale. The skill md file IS the description — the agent reads it lazily when the question touches that integration. One system-prompt sentence teaches the convention. | +| 7d | Skill md is the universal contract (sync-engine flavor OR api-only flavor) | Every connected integration gets a `skills/integrations/.md`. The skill's frontmatter `type` says which flavor. Idempotency check uses skill-existence. One unified surface for the main agent regardless of how rich the underlying sync is. | +| 8 | Sync agent transcripts persist in a dedicated `"system"` thread | Sync runs need debuggable transcripts (same as `spawn_sub_agent`) but don't belong in any user chat. v1: kernel creates a `"system"` thread lazily and writes there. v2: surface via `/system` browse command. | +| 9 | `runSubAgent` helper extracted from `spawn_sub_agent`'s body; reused by both `spawn_sub_agent` tool AND the queue consumer | Different system prompt + curated tool surface for each caller; only one piece of shared machinery (run/message persistence, abort cascade, token bookkeeping). | +| 10 | Dropbox is the v1 e2e smoke target despite being archetype B | Matches the user's VC-example demo. Channel + cursor + empty-ping primitives are needed in F1/F2 anyway since Dropbox uses them. | +| 11 | 3-PR stack (not 4) | Foundation primitives + queue trigger + agent + dropbox e2e split as F1/F2; F3 covers the other archetypes (Linear stable, gcal channel, Notion poll-or-webhook-bus). | +| 12 | Master spec + per-PR plans | One coherent architecture doc; per-PR plans land when each PR kicks off so they reflect the implementation reality. | From 90dd8a375c87d74c2135058be0eb1fa40e299f27 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 09:39:46 +0530 Subject: [PATCH 02/13] =?UTF-8?q?docs(sync-agent):=20drop=20handler.backfi?= =?UTF-8?q?ll=20export=20=E2=80=94=20backfill=20runs=20via=20run=5Fin=5Fsy?= =?UTF-8?q?nc?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per sauna pattern (src/agent/tools/write/run-in-sync.ts), backfill is not a persistent handler export. It runs as one-shot agent-authored ES-module code through a separate Worker Loader bundle, sharing the run_in_sync exec runner that also handles diagnostics, data repair, and probe-loop triggers. - §3.6: removed backfill from required exports; added explanatory note - §3.7: removed backfill from wrapper RPC list - §3.9: create_sync_engine no longer awaits handler.backfill - §3.10.3: split install (Phase 5) from backfill (Phase 5.5 via run_in_sync) - §3.10.4: create_sync_engine + run_in_sync descriptions updated - §4.1: trace shows run_in_sync exec bundle instead of facet.backfill() - §5.1: 'Backfill throws' reframed as run_in_sync return-value failure - §6 PR-F1: added exec-runner.ts + dropbox-backfill-exec.js fixture Co-Authored-By: Claude Opus 4.7 (1M context) --- .../specs/2026-05-17-sync-agent-design.md | 63 ++++++++++++++----- 1 file changed, 49 insertions(+), 14 deletions(-) diff --git a/docs/superpowers/specs/2026-05-17-sync-agent-design.md b/docs/superpowers/specs/2026-05-17-sync-agent-design.md index 928793b..0c79c17 100644 --- a/docs/superpowers/specs/2026-05-17-sync-agent-design.md +++ b/docs/superpowers/specs/2026-05-17-sync-agent-design.md @@ -419,9 +419,6 @@ The sync agent writes a single ES module string and stores it in `sync_engine.ha **v1 exports (the agent provides any subset that fits the integration):** ```ts -// Required for all integrations -export async function backfill(ctx: HandlerCtx): Promise; - // For webhook integrations (Archetypes A and B) export async function verifySignature(body: string, headers: Record, secret: string): Promise; export async function handleWebhook(req: Request, ctx: HandlerCtx): Promise; @@ -435,6 +432,8 @@ export async function renewWebhook(ctx: HandlerCtx, args: { id: string, label: s export async function pollOnce(ctx: HandlerCtx): Promise; ``` +**No `backfill` export.** Initial backfill is not part of the persistent handler module. The sync agent calls `run_in_sync({sync_id, code})` after `create_sync_engine` returns, passing one-shot ES-module source that walks the upstream API, writes the mirror via the same RPC surface, and populates the cursor (sauna pattern, `src/agent/tools/write/run-in-sync.ts`). Same surface is also used for diagnostics, data repair, and triggering upstream events during the probe loop — initial backfill is just the first use case. Worker Loader compiles the exec module separately from the handler bundle (different `codeId`), so backfill code doesn't get baked into the long-lived handler. + **`HandlerCtx` (supervisor-injected, minimal surface):** ```ts @@ -489,7 +488,7 @@ Each sync runs inside its own DO facet, created on demand from the Kernel DO sup The handler module the agent writes is wrapped by a supervisor-authored facet entrypoint (modeled on sauna's `src/facet/wrapper.ts`). The wrapper exports a class extending `DurableObject` (which is what `getDurableObjectClass("App")` returns from a Worker Loader bundle) that: -- Has one RPC method per agent-authored export (`verifySignature`, `handleWebhook`, `registerWebhook`, `unregisterWebhook`, `renewWebhook`, `pollOnce`, `backfill`). +- Has one RPC method per agent-authored handler export (`verifySignature`, `handleWebhook`, `registerWebhook`, `unregisterWebhook`, `renewWebhook`, `pollOnce`). No `backfill` RPC — backfill is run via the separate `run_in_sync` exec-runner path (§3.10.4), not the persistent handler bundle. - Each RPC method forwards args to the user module's named export, after injecting `HandlerCtx`. - **Bootstraps the facet's SQLite tables on first invocation** by running `sync_engine.schema_ddl` against `ctx.storage.sql`. Uses `CREATE TABLE IF NOT EXISTS` semantics so it's idempotent across facet wake-ups. Run once per facet lifecycle (the DDL itself is `IF NOT EXISTS`-guarded so re-runs are no-ops). On `update_sync`-bumped version, the new DDL runs against the existing tables — agent must include `CREATE TABLE IF NOT EXISTS` + `ALTER TABLE` migrations as needed (sauna pattern: `migration_ddl` is part of `update_sync`'s contract). - **Does NOT implement `alarm()`** — facets cannot use the alarm API (empirically validated; see §3.9). Renewal and poll dispatches arrive as ordinary RPC calls from the supervisor when its own alarm fires. @@ -580,7 +579,7 @@ So the supervisor (Kernel DO) owns the single alarm. Facets receive renewal/poll The `IDLE_TICK_MS` ceiling (5 min) guarantees the supervisor re-evaluates periodically even when no syncs are scheduled — picks up newly-created syncs without needing a wake-up. 4. `this.ctx.storage.setAlarm(nextAlarm)`. -**On `create_sync_engine`:** after the facet is bootstrapped and `backfill` returns, the supervisor calls `recomputeNextAlarm()` which calls `setAlarm` with the new minimum. A `webhook_channel` sync's first renewal gets scheduled immediately; a `poll` sync's first poll gets scheduled `poll_interval_sec` from now. +**On `create_sync_engine`:** after the facet is bootstrapped and `registerWebhook` returns (if applicable), the supervisor calls `recomputeNextAlarm()` which calls `setAlarm` with the new minimum. A `webhook_channel` sync's first renewal gets scheduled immediately; a `poll` sync's first poll gets scheduled `poll_interval_sec` from now. Initial backfill is the sync agent's next step — invoked separately via `run_in_sync` (§3.10.3 Phase 5.5) — and is not part of `create_sync_engine`'s critical path. **On `update_sync` changing strategy or intervals:** after the facet is re-instantiated with the new code, supervisor calls `recomputeNextAlarm()` again. @@ -766,7 +765,11 @@ Pick the strategy (`webhook_stable` / `webhook_channel` / `poll`) based on integ #### Phase 5 — Install (Branch A) -Call `create_sync_engine({integration, accountId, strategy, handlerJs, schemaDoc, skillPath, mirrorPath, pollIntervalSec?, renewalIntervalSec?})`. The kernel: compiles via Worker Loader (rolls back on compile failure), calls `handler.registerWebhook` if applicable, calls `handler.backfill` to populate the structural index, sets the initial alarm if needed. +Call `create_sync_engine({integration, accountId, strategy, handlerJs, schemaDoc, skillPath, mirrorPath, pollIntervalSec?, renewalIntervalSec?})`. The kernel: compiles via Worker Loader (rolls back on compile failure), calls `handler.registerWebhook` if applicable, sets the initial alarm if needed. Backfill is NOT part of this call. + +#### Phase 5.5 — Backfill (Branch A) + +Call `run_in_sync({sync_id, code})` with an ES-module string that walks the upstream API to populate the structural index. The exec runner gives the code `facet` (with `__query` / `__sqlExec` for the facet's own SQLite, and `__readMirror` / `__writeMirror` for the R2 mirror) plus `fetch` (Pipedream-proxied, same `X-Pd-App` rules as the handler). For long backfills, persist progress (cursor, `last_synced_at`) in either the mirror's `_cursor` field or a side table and call `run_in_sync` repeatedly — single calls are capped at ~30s wall-clock per Workers RPC. Backfill code is one-shot: it does NOT get baked into the persistent handler bundle. After the final call, write the cursor so the handler's `handleWebhook`/`pollOnce` can pick up from there. #### Phase 6 — Probe loop (Branch A; only for unfamiliar webhook shapes) @@ -792,9 +795,9 @@ Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`: **Write tools (all of these mutate the kernel DO or R2):** - `write_skill({slug, body})` — writes `skills/integrations/.md`. Validates frontmatter. -- `create_sync_engine({integration, account_id, strategy, handler_js, schema_ddl, schema_doc, skill_path, mirror_path, poll_interval_sec?, renewal_interval_sec?})` — inserts the row in status='discovering', compiles the bundle via Worker Loader (rolls back on compile error), boots the facet (which runs `schema_ddl` against the facet's SQLite), calls `handler.registerWebhook` if applicable, calls `handler.backfill`, sets first alarm. On success flips status to 'active'. On any failure: status='errored', errorText populated, returns the error. +- `create_sync_engine({integration, account_id, strategy, handler_js, schema_ddl, schema_doc, skill_path, mirror_path, poll_interval_sec?, renewal_interval_sec?})` — inserts the row in status='discovering', compiles the bundle via Worker Loader (rolls back on compile error), boots the facet (which runs `schema_ddl` against the facet's SQLite), calls `handler.registerWebhook` if applicable, sets first alarm. On success flips status to 'active'. On any failure: status='errored', errorText populated, returns the error. **Does NOT call backfill** — backfill is the agent's next step via `run_in_sync` (Phase 5.5). - `update_sync({sync_id, handler_js?, schema_ddl?, migration_ddl?, schema_doc?, strategy?, poll_interval_sec?, renewal_interval_sec?})` — atomic update. Bumps `version`. If `handler_js` changes, recompiles. If `schema_ddl` changes, runs `migration_ddl` against the existing facet SQLite BEFORE swapping (sauna pattern; agent provides the migration as `ALTER TABLE` / data-move statements). On migration failure, rolls back the whole update. If strategy changes from `poll` → `webhook_*`, registers webhook; vice versa unregisters. -- `run_in_sync({sync_id, code})` — one-shot execution of agent-authored JS inside the handler runtime. Same ctx as handlers. Used for backfill, diagnostics, triggering upstream actions during the probe loop. Output (return value + console logs) returned to the agent. +- `run_in_sync({sync_id, code})` — one-shot execution of agent-authored ES-module JS against a sync's facet (sauna pattern: separate Worker Loader bundle per call, not the persistent handler bundle). Code must export `async function runInSync(facet, fetch)` and return whatever the agent wants surfaced. `facet` exposes `__query(sql)` (read-only) and `__sqlExec(sql, ...bindings)` (writes/DDL) via the FacetBridge RpcTarget (raw DO stubs can't cross Worker Loader isolate boundaries — see sauna's run-in-sync.ts comment), plus `__readMirror()` / `__writeMirror(json)` for R2. `fetch` is Pipedream-proxied with the same `X-Pd-App` requirement as the handler. **This is the universal write-side channel for one-shot work**: initial backfill (Phase 5.5), diagnostics (read-only state inspection), data repair after a buggy handler version, and probe-loop triggers (firing a test event upstream so a webhook fires). Output capped at 64KB serialized JSON; logs from `console.log`/`info`/`warn`/`error` captured separately. Approval-gated. - `delete_sync({sync_id})` — destructive. Calls `handler.unregisterWebhook` if applicable, deletes `skills/integrations/.md` from R2, deletes `integrations//struct.json`, calls `ctx.facets.delete(sync.id)` (which drops the facet's SQLite — `raw_events`, `agent_dedup`, everything goes), flips sync_engine row to status='deleted' (kept for audit). For Branch-B (api-only) syncs there's no sync_engine row to flip — `delete_sync` just removes the skill. - `manage_tasks` — existing. The sync agent uses it to track its own phases (good UX: "Phase 1: discovery 🔄 Phase 2: schema 🔄 ..."). @@ -852,7 +855,7 @@ SyncAgent: (decide on flat-entries shape; write schema_doc) SyncAgent: manage_tasks update "Sync dropbox: Skill" SyncAgent: write_skill { slug: "dropbox", body: "" } SyncAgent: manage_tasks update "Sync dropbox: Handler" -SyncAgent: (write handler_js: verifySignature, handleWebhook (calls list_folder/continue with cursor), registerWebhook, unregisterWebhook, backfill) +SyncAgent: (write handler_js: verifySignature, handleWebhook (calls list_folder/continue with cursor), registerWebhook, unregisterWebhook — no backfill export) SyncAgent: create_sync_engine { integration: "dropbox", account_id: "acc_xyz", @@ -870,10 +873,40 @@ Kernel: recomputeNextAlarm() // supervisor recalculates the min next-a Kernel: await facet.registerWebhook({callbackUrl: "https:///webhook/dropbox/acc_xyz", label: "agent-os dropbox sync"}) → returns {id: "dbid_123", secret: "...", expiresAt: null} Kernel: UPDATE sync_engine SET webhook_id, webhook_secret, webhook_expires_at -Kernel: await facet.backfill() // facet runs backfill; writes mirror via MirrorIO RPC -Kernel: UPDATE sync_engine SET status='active', last_synced_at - (Facet wrapper's constructor already called setAlarm for renewal/poll if applicable. Dropbox: no alarm.) +Kernel: UPDATE sync_engine SET status='active' + (Supervisor recomputeNextAlarm() — dropbox has no expiry → no alarm contribution.) ↓ +SyncAgent: manage_tasks update "Sync dropbox: Backfill" +SyncAgent: run_in_sync { + sync_id, + code: "export async function runInSync(facet, fetch) { + const entries = []; + let cursor = null; + // First page + let res = await fetch('https://api.dropboxapi.com/2/files/list_folder', { + method: 'POST', + headers: { 'X-Pd-App': 'dropbox', 'Content-Type': 'application/json' }, + body: JSON.stringify({ path: '', recursive: true }) + }).then(r => r.json()); + entries.push(...res.entries); + cursor = res.cursor; + while (res.has_more) { + res = await fetch('https://api.dropboxapi.com/2/files/list_folder/continue', { + method: 'POST', + headers: { 'X-Pd-App': 'dropbox', 'Content-Type': 'application/json' }, + body: JSON.stringify({ cursor }) + }).then(r => r.json()); + entries.push(...res.entries); + cursor = res.cursor; + } + await facet.__writeMirror({ _cursor: cursor, entries }); + return { entries: entries.length }; + }" + } +Kernel: load exec bundle via env.LOADER.load(buildExecWorkerCode({ code, ... })) — separate from the handler bundle + → calls FacetBridge.runInSync(facet, fetch) + → exec returns { ok: true, result: { entries: 1247 }, logs: [...] } +SyncAgent: (sees {entries: 1247}; mirror is now populated; cursor is in _cursor) SyncAgent: manage_tasks update "Sync dropbox: Verify" SyncAgent: read_mirror sync_id → confirms shape SyncAgent: get_sync_errors sync_id → empty @@ -982,7 +1015,7 @@ exec_code returns success. - **Discovery times out / hits cap.** Sync agent writes what it has, marks the sync engine with `status='errored'`, errorText explains. User can `/sync rerun dropbox` to retry (forces re-publish of the connection event, bypassing idempotency). - **Handler compile failure (Worker Loader rejects the bundle).** Caught inside `create_sync_engine` BEFORE facet construction completes — supervisor calls `ctx.facets.delete(sync.id)` to ensure no partial facet exists, returns the compile error to the agent inline. The agent fixes the handler and re-calls. - **registerWebhook fails (upstream rejects).** `facet.registerWebhook` throws → supervisor catches → marks status='errored' with the error text → cleans up via `ctx.facets.delete(sync.id)` → returns error to agent. Agent diagnoses (wrong scope, wrong app slug, malformed body), edits handler, re-calls `create_sync_engine`. -- **Backfill throws.** Same shape — but facet stays alive (webhook is already registered upstream). Agent uses `run_in_sync` or `update_sync` to fix and retry. Supervisor doesn't auto-delete on backfill failure because the upstream-registered webhook would need to be unregistered first, which is expensive. +- **Backfill (`run_in_sync`) throws.** sync_engine is already 'active' and the webhook is already registered upstream — the throw is contained in the exec runner's `{ ok: false, error, detail, stack, logs }` return value (sauna pattern). Agent reads the logs, edits the backfill code, calls `run_in_sync` again. Supervisor does not auto-delete on backfill failure because the upstream-registered webhook would need to be unregistered first, which is expensive — and partial mirrors are fine: incremental webhook deliveries can fill the gap. If the agent gives up, the user can `/sync delete ` and `/sync rerun `. ### 5.2 At webhook delivery - **verifySignature returns false.** INSERT `sync_errors{error_message:"signature verification rejected", payload_preview}`, return 401. NO sync state change. This is normal scanner noise on a public webhook URL — only worrying if it persists alongside zero successful deliveries (see sauna's debugging rubric, ported into the sync-agent prompt). The rubric: compare `count(sync_errors WHERE occurred_at > T)` against `sync_engine.last_synced_at` — if errors are high AND last_synced_at is stale, the verifier is broken; if errors are high BUT last_synced_at is fresh, it's just scanner noise. @@ -1009,7 +1042,9 @@ exec_code returns success. - `apps/kernel/src/sync/mirror-io.ts` — `MirrorIO` WorkerEntrypoint (R2 read/write scoped by `props.mirrorPath`) - `apps/kernel/src/sync/dispatch.ts` — `dispatchWebhook` body (looks up sync row, calls `getOrCreateFacet`, RPCs verifySignature + handleWebhook) - `apps/cli/src/commands/sync.ts` — `/sync status` and `/sync delete` (no `/sync ` trigger; queue is the trigger) -- `__tests__/fixtures/dropbox-handler.js` — hand-written smoke fixture (verifySignature, handleWebhook with cursor + list_folder/continue, registerWebhook, unregisterWebhook, backfill) +- `__tests__/fixtures/dropbox-handler.js` — hand-written smoke fixture (verifySignature, handleWebhook with cursor + list_folder/continue, registerWebhook, unregisterWebhook). No backfill export — that runs via the exec runner. +- `apps/kernel/src/sync/exec-runner.ts` — `buildExecWorkerCode({code, syncId, ...})` + `FacetBridge` RpcTarget (mirrors `__query`, `__sqlExec`, `__readMirror`, `__writeMirror` so agent code can talk to the facet across the Worker Loader isolate boundary — sauna's `src/execrunner/*`) +- `__tests__/fixtures/dropbox-backfill-exec.js` — hand-written smoke fixture for the backfill path (`runInSync(facet, fetch)` that walks list_folder, writes mirror) **Files modified:** - `apps/kernel/src/index.ts` — add `POST /webhook/:integration/:account_id` route; export `HttpGateway` and `MirrorIO` WorkerEntrypoints alongside existing PipedreamProxy etc. From 2045338dc4aac6715e0ae772dc6886ab26a07e90 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 09:52:23 +0530 Subject: [PATCH 03/13] docs(sync-agent): mirror is a prefix-scoped KV space, not a single file MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit User feedback: forcing one struct.json per sync over-constrains integrations that benefit from multi-file layouts (Gmail: labels.json + cursors.json + threads/by-label/*.json; Drive: per-folder index). Agent should design the layout; the supervisor just provides scoped FS primitives. Replaces MirrorIO (readMirror/writeMirror unknown-typed) with MirrorFS exposing read/write/list/delete over R2 keys relative to a per-sync prefix. - §3.3: mirror_path column → mirror_prefix (defaults integrations//) - §3.4: rewritten — agent designs file layout; Dropbox/Gmail/Airtable examples - §3.5: Dropbox skill example updated (mirror_prefix + Mirror layout section) - §3.6: HandlerCtx.readMirror/writeMirror → ctx.mirror.{read,write,list,delete} - §3.7: MirrorIO → MirrorFS; capabilities pass-through carries mirrorPrefix - §3.10.4: read_mirror takes a path; add list_mirror; run_in_sync FacetBridge exposes __read/__write/__list/__delete instead of __readMirror/__writeMirror - §3.10.4: create_sync_engine/delete_sync use mirror_prefix; delete sweeps every R2 key under the prefix (batch delete) - §4.1/§4.2 traces: handler/exec code uses ctx.mirror.read("tree.json") + facet.__write("tree.json", ...) shape - §4.3 main-agent trace: read_file integrations/dropbox/tree.json - §6 PR-F1: mirror-io.ts → mirror-fs.ts; exec-runner FacetBridge updated - §10 decisions: row 6 reframed (cursor location is agent's call); added row 6c (mirror is prefix-scoped KV space, not single file — rationale) Co-Authored-By: Claude Opus 4.7 (1M context) --- .../specs/2026-05-17-sync-agent-design.md | 245 ++++++++++-------- 1 file changed, 134 insertions(+), 111 deletions(-) diff --git a/docs/superpowers/specs/2026-05-17-sync-agent-design.md b/docs/superpowers/specs/2026-05-17-sync-agent-design.md index 0c79c17..a598fe7 100644 --- a/docs/superpowers/specs/2026-05-17-sync-agent-design.md +++ b/docs/superpowers/specs/2026-05-17-sync-agent-design.md @@ -17,7 +17,7 @@ When a user connects a third-party integration via the existing `/connect` flow, - **Sync-engine flavor** (for sync-able integrations): explains the mirror's shape, query patterns over the mirror, and `exec_code` snippets for fetching content / writing upstream. - **API-only flavor** (for everything else): explains the API surface with `exec_code` snippets. No mirror, no sync engine. -3. **Optionally installs a sync engine** — handler JS code, mirror json shape, webhook (or polling) wiring — but only when triage said it's worth it. Many integrations get just a skill and exit; no `sync_engine` row, no facet, no mirror, no webhook. +3. **Optionally installs a sync engine** — handler JS code, mirror layout (one or more R2 files under a per-sync prefix), webhook (or polling) wiring — but only when triage said it's worth it. Many integrations get just a skill and exit; no `sync_engine` row, no facet, no mirror, no webhook. 4. Returns control. The main agent from then on uses the integration by reading its skill: for sync-able integrations it grep-reads the mirror for navigation then `exec_code`s for content/writes; for API-only integrations it just `exec_code`s. @@ -54,7 +54,7 @@ This is sauna-shaped, with two deliberate departures and one inherited primitive - **Inherited from sauna: per-sync Cloudflare DO Facets.** Each sync runs inside its own DO facet (`ctx.facets.get(sync.id, ...)`). The facet has isolated SQLite (handler scratch space — dedup tables, ephemeral state), receives capabilities from the Kernel DO supervisor via RPC, and self-schedules its own renewal/poll alarms. Lifecycle primitives (`ctx.facets.abort()` on update, `ctx.facets.delete()` on delete) map cleanly to `update_sync` / `delete_sync`. Facets are not separately billed. - **Departure 1: No `ctx.agent` in handlers.** Sync handlers stay inline JS — no runtime LLM reasoning. -- **Departure 2: Mirror is R2 json, not facet SQLite.** Per user direction. The main agent grep-reads `integrations//struct.json` with the existing `read_file` tool — no new query primitive. The facet's SQLite stays available for handler scratch (cursors, dedup) but is not the main-agent-facing data layer. +- **Departure 2: Mirror is a prefix-scoped R2 key-value space, not facet SQLite and not a single file.** Per user direction. Each sync owns `integrations//` (or a custom prefix); the agent decides the layout — one file for simple integrations, many files for those where partial updates / locality matter (§3.4). The main agent reads what it needs with the existing `read_file` tool — no new query primitive. The facet's SQLite stays available for handler scratch (cursors, dedup) but is not the main-agent-facing data layer. - **Departure 3: No `api_hosts` allowlist.** Every fetch from the handler runtime and from `exec_code` goes through the Pipedream proxy, which enforces the `X-Pd-App` slug ↔ host pairing itself. No additional egress gate. --- @@ -83,7 +83,7 @@ The main agent's runtime flow for a question that touches a connected integratio 1. **Sees** the integration's slug in ``. 2. **Reads** `skills/integrations/.md` via the existing `read_file` tool. 3. **Branches** on what the skill says: - - Sync-engine skill: skill points to `integrations//struct.json` and describes its shape. Agent `read_file`s the mirror, greps inline, finds what it needs, then `exec_code`s the upstream API (with `X-Pd-App: `) for actual content / writes. + - Sync-engine skill: skill describes the mirror layout under `integrations//` — which files exist, what shape each has, what queries each answers. Agent `read_file`s the right mirror file (or `list`s the prefix first if the layout is multi-file), greps inline, finds what it needs, then `exec_code`s the upstream API (with `X-Pd-App: `) for actual content / writes. - API-only skill: skill describes the API surface only. Agent goes straight to `exec_code`. 4. **No per-integration tool wrappers exist** — `exec_code` is the universal surface for both reads and writes against the upstream. @@ -115,9 +115,9 @@ export const syncEngine = sqliteTable("sync_engine", { // Authoring artifacts (sync-agent-produced) handlerJs: text("handler_js").notNull(), // ES module source string schemaDdl: text("schema_ddl").notNull().default(""), // SQL DDL for the facet's own SQLite (dedup tables, raw_events for probe loop, cursor state, etc.). Empty string allowed for handlers that don't need any SQL state. Run by the supervisor at facet bootstrap. - schemaDoc: text("schema_doc").notNull(), // free-form prose describing the mirror json shape (R2) + schemaDoc: text("schema_doc").notNull(), // free-form prose describing the mirror layout: which files live under the prefix, what shape each one has, what's in them vs. fetched on demand skillPath: text("skill_path").notNull(), // "skills/integrations/dropbox.md" - mirrorPath: text("mirror_path").notNull(), // "integrations/dropbox/struct.json" + mirrorPrefix: text("mirror_prefix").notNull(), // R2 key prefix, e.g. "integrations/dropbox/" — defaults to `integrations//`. Every mirror read/write/list/delete from the facet is scoped under this prefix. The agent decides whether the layout is one file (tree.json) or many (labels.json, threads/by-label/inbox.json, ...); the supervisor never inspects the contents. // Webhook state webhookId: text("webhook_id"), // upstream's id for the registered webhook (for unregister) webhookSecret: text("webhook_secret"), // for verifySignature @@ -165,69 +165,73 @@ Only inserted when something throws: DESC index on `(sync_id, occurred_at)` supports `SELECT ... ORDER BY occurred_at DESC LIMIT N` efficiently — that's the `get_sync_errors` read pattern. Cascade-deletes when the parent sync row is removed. -The cursor (when an integration needs one) lives **inside the mirror json itself** under a reserved `_cursor` field. Reasoning: simpler than a separate column or sidecar file, atomic with the mirror write, and main-agent code that greps the mirror can simply ignore `_cursor`. +Cursors live in the mirror — but the agent decides *where* in the mirror. Could be a reserved `_cursor` field in the single tree file, a sidecar `cursors.json`, a per-resource map (e.g. `cursors/.json`), or a row in the facet's SQLite if the cursor is high-churn. Whatever fits the integration. The supervisor doesn't care. -### 3.4 The mirror — `integrations//struct.json` in R2 +### 3.4 The mirror — a prefix-scoped key-value space in R2 -Path: `integrations//struct.json` (per deployment — the Kernel DO is single-tenant). +The mirror is a **filesystem-style prefix in R2**, not a single file. Each sync gets a prefix (`mirror_prefix` column, defaults to `integrations//`) and reads/writes/lists/deletes keys underneath it. The agent decides the file layout — one file, many files, nested prefixes, whatever makes the integration navigable. -Shape: **agent-authored at `create_sync_engine` time**, free-form within the constraint that it's a valid json document AND adheres to the index-not-content principle from §1.1. The agent's job is to pick the smallest shape that makes the integration efficiently navigable + queryable. Concretely: +Primitives (exposed to the handler via `ctx.mirror` and to one-shot exec code via `facet.__read/__write/__list/__delete`): -**Dropbox (file tree integration) — agent will naturally land on:** - -```json -{ - "_cursor": "AAH9p3...", - "_updated_at": "2026-05-17T18:31:01Z", - "entries": [ - { "id": "id:abc", "path": "/Personal/Tax/2024-form-1040.pdf", "name": "2024-form-1040.pdf", "type": "file", "size": 142331, "modified": "2026-04-15T03:21:00Z", "mime": "application/pdf" }, - { "id": "id:def", "path": "/Fund/Dimension-I/Portfolio/AcmeCorp/one-pager.md", "name": "one-pager.md", "type": "file", "size": 3104, "modified": "2026-05-12T11:02:00Z", "mime": "text/markdown" }, - ... - ] +```ts +interface MirrorFS { + read(path: string): Promise; // string body or null if absent; agent chooses encoding (JSON, ndjson, markdown, plain text) + write(path: string, body: string): Promise; // overwrite-by-default; no conditional headers in v1 + list(prefix?: string): Promise; // returns keys RELATIVE to mirror_prefix; pass "" or omit for all + delete(path: string): Promise; } ``` +All `path` arguments are **relative to `mirror_prefix`** — the supervisor's `MirrorFS` WorkerEntrypoint prepends the prefix before hitting R2. Paths can never escape (no `..`, no leading `/`); the entrypoint rejects. + +**Why prefix-scoped instead of a single file:** + +- Concurrency: webhook handlers can update one file (e.g. a per-label index) without rewriting the whole tree. Single-file mirrors force whole-tree rewrites on every delta and serialize writes through R2's strong-consistency layer. +- Size: a Gmail mirror that flattens everything into one json bloats fast. Splitting by label / by-page keeps each file small enough to grep with `read_file` without overflowing the main agent's context window. +- Locality: main agent finds what it needs faster when the layout itself encodes structure. Reading `integrations/gmail/threads/by-label/inbox.json` is faster than greping a giant `gmail.json` for `label: "INBOX"`. +- Cost: R2 charges per operation. Partial updates win when only a slice of the index changes. + +**The agent designs the layout and documents it in the skill md.** That layout *is* the schema_doc. Examples: + +**Dropbox (file tree) — single file is fine.** Whole tree is small, all-or-nothing replacement on each empty-ping delta is cheap. + +``` +integrations/dropbox/ +└── tree.json # { _cursor, _updated_at, entries: [{id, path, name, type, size, modified, mime}, ...] } +``` + `schema_doc`: -> Flat entries array. Each entry: `{id, path, name, type, size, modified, mime}`. `type` is "file" or "folder". Sorted by `path` ascending. `_cursor` is Dropbox's list_folder/continue cursor — handler reads, fetches deltas, applies, saves new cursor. File contents NOT mirrored; download via `https://content.dropboxapi.com/2/files/download` with `X-Pd-App: dropbox` from `exec_code`. - -**Airtable (database integration) — different shape entirely:** - -```json -{ - "_cursor": "...", - "_updated_at": "2026-05-17T18:31:01Z", - "bases": [ - { "id": "appXyz", "name": "Investor Pipeline" } - ], - "tables": [ - { - "base_id": "appXyz", - "id": "tblAbc", - "name": "Portfolio Companies", - "primary_field_id": "fldName", - "fields": [ - { "id": "fldName", "name": "Company Name", "type": "singleLineText" }, - { "id": "fldStage", "name": "Stage", "type": "singleSelect", "options": ["Seed", "Series A", "Series B", "Series C"] }, - { "id": "fldCheck", "name": "Check Size", "type": "currency" }, - { "id": "fldNotes", "name": "Notes", "type": "longText" } - ] - }, - { - "base_id": "appXyz", - "id": "tblDef", - "name": "Deal Flow", - "fields": [ ... ] - } - ] -} +> Single `tree.json` at the prefix root. `entries` is a flat array of `{id, path, name, type, size, modified, mime}`. `_cursor` is Dropbox's list_folder/continue cursor. File contents NOT mirrored; download via `https://content.dropboxapi.com/2/files/download` with `X-Pd-App: dropbox` from `exec_code`. + +**Gmail (high-volume, multiple useful slices) — multi-file pays off.** + +``` +integrations/gmail/ +├── labels.json # { labels: [{id, name, type, ...}] } — small, queried often +├── cursors.json # { historyId, perLabelLastSeen } — separated so handler can rewrite without touching labels +└── threads/ + └── by-label/ + ├── INBOX.json # { threads: [{id, snippet, from, subject, date, hasAttachments}, ...] } — last N threads per label + ├── SENT.json + └── Label_2742.json +``` + +`schema_doc`: +> `labels.json` is the full label list. `cursors.json` tracks Gmail History API `historyId` plus per-label last-seen markers. `threads/by-label/.json` holds the most recent 200 thread headers per label (id, snippet, from, subject, date, hasAttachments) — full bodies fetched on demand via Gmail API. Main agent finds threads by reading the right label file directly (e.g. `read_file integrations/gmail/threads/by-label/INBOX.json`); doesn't need to grep a global thread list. + +**Airtable (schema-of-schemas) — single file again; schema is small.** + +``` +integrations/airtable/ +└── workspace.json # { _cursor, bases: [...], tables: [...] } ``` `schema_doc`: -> Index of the user's Airtable workspace. `bases` lists workspaces, `tables` lists tables within each base with their full field schema (incl. select options for enums). Records NOT mirrored — main agent fetches them via Airtable API on demand using the field IDs from this index. +> Single workspace file. `bases` lists workspaces, `tables` lists tables within each base with their full field schema (incl. select options for enums). Records NOT mirrored — main agent fetches them via Airtable API on demand using the field IDs from this index. -**The principle in action:** the agent does NOT enumerate every record in the Airtable. It enumerates bases + tables + fields. That's the "menu" the main agent reads to compose a runtime API call (`exec_code` → `https://api.airtable.com/v0/appXyz/tblAbc?filterByFormula=...`). Same idea for Linear (mirror = team list + project list + issue headers; NOT issue descriptions), GitHub (mirror = repo list + issue/PR headers; NOT file contents), and so on. +**The principle in action:** Dropbox's tree is one file because the tree is small and any change wants the whole thing. Gmail's threads-by-label is many files because labels change independently, individual label files stay small, and the layout itself is a hint to the main agent. Airtable's schema is one file because the schema is bounded. Same Prime Directive (§1.1) — **mirror is structural index, not content** — but the file count is integration-specific. -This schema_doc gets included in the structure-skill so the main agent knows the shape without having to infer it. +Whatever the agent picks, the skill md (§3.5) MUST describe the layout so the main agent knows where to look. ### 3.5 The integration skill — `skills/integrations/.md` in R2 @@ -245,14 +249,17 @@ description: Folder layout, naming conventions, and query patterns for this user type: sync-engine integration: dropbox sync_id: 01HW...XYZ -mirror_path: integrations/dropbox/struct.json +mirror_prefix: integrations/dropbox/ skill_version: 1 generated_at: 2026-05-17T18:30:12Z --- # Dropbox structure -The mirror at `integrations/dropbox/struct.json` is a flat entries array... +## Mirror layout + +Single file: `integrations/dropbox/tree.json`. Shape: `{ _cursor, _updated_at, entries: [{id, path, name, type, size, modified, mime}, ...] }`. See schema_doc below for full detail. + [schema_doc embedded here] ## Folder layout @@ -274,13 +281,13 @@ Everything lives under two roots: ## What's in the mirror vs. fetched on demand -In the mirror (`integrations/dropbox/struct.json`): file/folder paths, names, ids, types, sizes, modified-at, mime-types. **NOT** file contents. +In the mirror (`integrations/dropbox/tree.json`): file/folder paths, names, ids, types, sizes, modified-at, mime-types. **NOT** file contents. To get a file's contents, `exec_code` the Dropbox download API (snippet under "Reading content" below). ## Query patterns (mirror-only — no API call needed) -- Find tax filings: read `integrations/dropbox/struct.json`, filter `entries` where `path` starts with `/Personal/Tax/` +- Find tax filings: read `integrations/dropbox/tree.json`, filter `entries` where `path` starts with `/Personal/Tax/` - Find a portfolio company's docs: filter `path` matches `/Fund//Portfolio//` - Find the most recent board deck for a company: filter `path` matches `/Fund/*/Portfolio//Board/`, sort by `modified` desc @@ -443,16 +450,21 @@ interface HandlerCtx { fetch: typeof fetch; // R2 mirror access — RPC back to the supervisor (which owns the R2 binding). - // Scoped to this sync's mirror_path; the facet can't read or write any - // other sync's mirror. - readMirror(): Promise; - writeMirror(j: unknown): Promise; + // Scoped to this sync's mirror_prefix; paths are RELATIVE to the prefix. + // The handler decides the file layout (one file, many files, nested + // directories — see §3.4). Cannot read or write outside its prefix. + mirror: { + read(path: string): Promise; + write(path: string, body: string): Promise; + list(prefix?: string): Promise; + delete(path: string): Promise; + }; // Facet's own SQLite — direct access. Tables defined by the sync agent's // schema_ddl, run by the supervisor at facet bootstrap (§3.7). Used for: // - agent_dedup tables (idempotency keys, sauna pattern) // - raw_events table during the probe loop (phase 1 capture) - // - cursor state when the agent prefers SQL over a field in the mirror json + // - cursor state when the agent prefers SQL over a mirror file // - retry/backoff bookkeeping // - any per-handler-invocation state the agent decides matters // NOT the main-agent-facing data layer (that's the r2 mirror). @@ -476,9 +488,11 @@ interface HandlerCtx { } ``` -That's it. No supervisor SQL access (only facet's own sqlite). No sub-agent spawning. No process or filesystem access beyond mirror RPCs. The handler is a pure async function with fetch + mirror read/write + scratch sqlite + self-alarm. +That's it. No supervisor SQL access (only facet's own sqlite). No sub-agent spawning. No process or filesystem access beyond mirror RPCs. The handler is a pure async function with fetch + mirror primitives + scratch sqlite. + +**Why this is enough for Dropbox:** the handler `mirror.read("tree.json")` (parses it to get `_cursor`), calls `https://api.dropboxapi.com/2/files/list_folder/continue` with the cursor, receives a list of file changes, applies them to `entries`, updates `_cursor`, `mirror.write("tree.json", JSON.stringify(next))`. ~30 lines of JS. -**Why this is enough for Dropbox:** the handler reads the current mirror (gets `_cursor`), calls `https://api.dropboxapi.com/2/files/list_folder/continue` with the cursor, receives a list of file changes, applies them to the entries array, updates `_cursor`, writes the mirror back. ~30 lines of JS. Cursor could also live in `ctx.sql` if the agent prefers — choice is the agent's. +**Why this is enough for Gmail:** handler receives a webhook for label X, `mirror.read("cursors.json")` to get the per-label cursor, fetches the history slice via the Gmail API, `mirror.read("threads/by-label/.json")`, applies the delta to that slice only, writes it back, updates `cursors.json`. The unchanged label files stay untouched. ### 3.7 Per-sync facet (Cloudflare DO Facets primitive) @@ -505,10 +519,13 @@ const facet = this.ctx.facets.get(sync.id, async () => { return { class: HandlerClass, // Capabilities the facet receives — facet code cannot construct these itself. - // sauna pattern: HttpGateway proxies the Pipedream-routed fetch; MirrorIO - // scopes r2 access to this sync's mirror_path only. + // sauna pattern: HttpGateway proxies the Pipedream-routed fetch. httpGateway: this.ctx.exports.HttpGateway({ props: { integration: sync.integration, accountId: sync.accountId } }), - mirrorIO: this.ctx.exports.MirrorIO({ props: { mirrorPath: sync.mirrorPath } }), + // MirrorFS scopes r2 access to this sync's mirror_prefix only. The handler + // calls mirror.read/write/list/delete with paths RELATIVE to the prefix; + // MirrorFS prepends the prefix and rejects `..` / leading `/` so a sync + // can never read or write another sync's mirror. + mirrorFS: this.ctx.exports.MirrorFS({ props: { mirrorPrefix: sync.mirrorPrefix } }), // Static metadata baked into wrapper as constants. syncId: sync.id, integration: sync.integration, @@ -521,7 +538,7 @@ const facet = this.ctx.facets.get(sync.id, async () => { }); ``` -`HttpGateway` and `MirrorIO` are `WorkerEntrypoint` classes exported by the Kernel Worker — the same pattern agent-os already uses for `PipedreamProxy`, `BrowserBridge`, `ExecRunner`, `Fs`. Each is a thin RPC surface that the facet code calls; the supervisor enforces auth + scoping inside the RPC methods. +`HttpGateway` and `MirrorFS` are `WorkerEntrypoint` classes exported by the Kernel Worker — the same pattern agent-os already uses for `PipedreamProxy`, `BrowserBridge`, `ExecRunner`, `Fs`. Each is a thin RPC surface that the facet code calls; the supervisor enforces auth + scoping inside the RPC methods. **Worker Loader caching:** @@ -616,7 +633,7 @@ Kernel DO: ensure "system" thread exists, create sync agent run row (kind="sync" ↓ Sync agent: Phase 1 — Triage. Is this integration sync-able (has structural surface worth mirroring)? - ├── YES → Phases 2–7 (schema, skill, handler, install, probe, verify). Writes skill md + struct.json, creates sync_engine row + facet. + ├── YES → Phases 2–7 (schema, skill, handler, install, probe, verify). Writes skill md + mirror file(s) under `integrations//`, creates sync_engine row + facet. └── NO → Phase 2-API. Writes a short API-only skill md (no sync_engine row, no facet, no mirror). Exits in ~30 seconds. ↓ On completion: queue ack. On failure: log + DLQ retry policy applies. @@ -626,7 +643,7 @@ On completion: queue ack. On failure: log + DLQ retry policy applies. **Why run on every connection rather than gating by integration type:** 3,000+ Pipedream integrations and counting. Maintaining a hand-curated allowlist of "sync-able" slugs would (a) rot fast, (b) miss new integrations, (c) can't reason about *this user's* specific use of a flexible integration like Notion (could be a wiki, could be a database, could be both). The agent is the only piece capable of looking at the actual integration and deciding. The cost of running the agent on a write-only API like MailChimp is ~30 seconds of one cheap turn — small enough to absorb. -**Why a "system" thread for the run:** sync agent runs need to persist their transcript for debuggability (run row + messages, same as `spawn_sub_agent`). They don't belong in any user chat thread — they weren't typed there. v1 routes them to a special thread with id `"system"` that the kernel creates lazily. v2 may expose this via `/system` command for browsing. For v1 the user doesn't see the transcript directly; only the resulting sync_engine row, skill md, and mirror json are user-facing. +**Why a "system" thread for the run:** sync agent runs need to persist their transcript for debuggability (run row + messages, same as `spawn_sub_agent`). They don't belong in any user chat thread — they weren't typed there. v1 routes them to a special thread with id `"system"` that the kernel creates lazily. v2 may expose this via `/system` command for browsing. For v1 the user doesn't see the transcript directly; only the resulting sync_engine row, skill md, and mirror files are user-facing. **Run shape:** same `runSubAgent` helper that `spawn_sub_agent` uses (extracted in PR-F2). Differences: - `systemPrompt`: `SYNC_AGENT_SYSTEM_PROMPT` (port of sauna's, adapted for json+md output) @@ -662,7 +679,7 @@ Heavy port of `sauna-assignment/src/agent/system-prompt.ts`, restructured for th **The two branches.** Every run takes exactly one path through the procedure: -- **Branch A — Sync engine (sync-able integrations).** File systems, databases, ticket trackers, wikis, calendars, messaging systems. The integration has a navigable structure that the user benefits from indexing locally. Phases 1 → 2 → 3 → 4 → 5 → 6 → 7. Outputs: skill md + struct.json + sync_engine row + active facet. +- **Branch A — Sync engine (sync-able integrations).** File systems, databases, ticket trackers, wikis, calendars, messaging systems. The integration has a navigable structure that the user benefits from indexing locally. Phases 1 → 2 → 3 → 4 → 5 → 6 → 7. Outputs: skill md + agent-designed mirror file(s) under `integrations//` + sync_engine row + active facet. - **Branch B — API-only (everything else).** Write-only APIs (MailChimp, Twilio, Sendgrid), AI APIs (OpenAI, Anthropic), payments (Stripe-write), social broadcasters (LinkedIn, WhatsApp), most read-it's-cheap APIs. Phase 1 → Phase 2-API → exit. Outputs: skill md only. #### Phase 1 — Triage (every run starts here) @@ -695,7 +712,7 @@ Write `skills/integrations/.md` with frontmatter `type: api-only` and a bo - Cross-references to the integration's official docs URL. - Caveats / scope. -Then exit. No `create_sync_engine` call. No struct.json. The skill alone is the deliverable. +Then exit. No `create_sync_engine` call. No mirror files. The skill alone is the deliverable. #### Phase 2 — Branch A: PRIME DIRECTIVE re-read + schema design @@ -716,7 +733,7 @@ THE PRIME DIRECTIVE — re-read before designing the schema: > > Sanity check: imagine the user has 50,000 records / 100GB of files / 10,000 issues. Does your mirror stay under ~5MB? If not, you're mirroring content. -Decide the mirror json shape based on what discovery surfaced, applying the Prime Directive ruthlessly. Write the `schema_doc` prose. The schema_doc must explicitly call out "what's in the mirror vs. fetched on demand" so the main agent knows both halves. +Decide the mirror layout based on what discovery surfaced, applying the Prime Directive ruthlessly: how many files, named what, with what shape (see §3.4 examples — Dropbox = single tree.json; Gmail = labels/cursors/per-label-threads). Write the `schema_doc` prose. The schema_doc must list every file the mirror will contain, the shape of each, AND explicitly call out "what's in the mirror vs. fetched on demand" so the main agent knows both halves. Then decide the **facet SQLite schema** — separate from the R2 mirror. Write `schema_ddl` as a SQL string with `CREATE TABLE IF NOT EXISTS` statements. This is the handler-internal state layer; not visible to the main agent. Typical contents: @@ -727,10 +744,12 @@ CREATE TABLE IF NOT EXISTS agent_dedup ( claimed_at INTEGER NOT NULL ); --- Optional: cursor in SQL if you'd rather not embed it in the mirror json. --- For Dropbox: leave cursor in the mirror's _cursor field — simpler. --- For more complex multi-cursor integrations (one cursor per Google Drive change-channel), --- a table is cleaner: +-- Optional: cursor in SQL if you'd rather not put it in a mirror file. +-- For Dropbox: leave cursor in tree.json's _cursor field — simpler. +-- For Gmail: cursors.json works fine (one file, low write churn). +-- For high-churn multi-cursor integrations (Google Drive: one cursor per +-- change-channel, updated on every webhook), facet SQLite is cleaner since +-- the cursor write doesn't touch R2 on every single webhook: CREATE TABLE IF NOT EXISTS cursors ( resource_id TEXT PRIMARY KEY, cursor TEXT NOT NULL, @@ -765,11 +784,11 @@ Pick the strategy (`webhook_stable` / `webhook_channel` / `poll`) based on integ #### Phase 5 — Install (Branch A) -Call `create_sync_engine({integration, accountId, strategy, handlerJs, schemaDoc, skillPath, mirrorPath, pollIntervalSec?, renewalIntervalSec?})`. The kernel: compiles via Worker Loader (rolls back on compile failure), calls `handler.registerWebhook` if applicable, sets the initial alarm if needed. Backfill is NOT part of this call. +Call `create_sync_engine({integration, accountId, strategy, handlerJs, schemaDoc, skillPath, mirrorPrefix?, pollIntervalSec?, renewalIntervalSec?})`. `mirrorPrefix` defaults to `integrations//` — agent only passes a custom value for unusual cases (multi-account isolation, manual namespacing). The kernel: compiles via Worker Loader (rolls back on compile failure), calls `handler.registerWebhook` if applicable, sets the initial alarm if needed. Backfill is NOT part of this call. #### Phase 5.5 — Backfill (Branch A) -Call `run_in_sync({sync_id, code})` with an ES-module string that walks the upstream API to populate the structural index. The exec runner gives the code `facet` (with `__query` / `__sqlExec` for the facet's own SQLite, and `__readMirror` / `__writeMirror` for the R2 mirror) plus `fetch` (Pipedream-proxied, same `X-Pd-App` rules as the handler). For long backfills, persist progress (cursor, `last_synced_at`) in either the mirror's `_cursor` field or a side table and call `run_in_sync` repeatedly — single calls are capped at ~30s wall-clock per Workers RPC. Backfill code is one-shot: it does NOT get baked into the persistent handler bundle. After the final call, write the cursor so the handler's `handleWebhook`/`pollOnce` can pick up from there. +Call `run_in_sync({sync_id, code})` with an ES-module string that walks the upstream API to populate the mirror. The exec runner gives the code `facet` (with `__query` / `__sqlExec` for the facet's own SQLite, and `__read(path)` / `__write(path, body)` / `__list(prefix?)` / `__delete(path)` for the R2 mirror — all paths relative to the sync's mirror_prefix) plus `fetch` (Pipedream-proxied, same `X-Pd-App` rules as the handler). For long backfills, persist progress (cursor, page tokens) somewhere in the mirror (the agent picks where — `cursors.json`, an embedded field in the main file, or a facet SQLite row) and call `run_in_sync` repeatedly — single calls are capped at ~30s wall-clock per Workers RPC. Backfill code is one-shot: it does NOT get baked into the persistent handler bundle. After the final call, write the final cursor so the handler's `handleWebhook`/`pollOnce` can pick up from there. #### Phase 6 — Probe loop (Branch A; only for unfamiliar webhook shapes) @@ -789,16 +808,17 @@ Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`: - `pipedream_fetch({host, path, method, headers, body, app_slug})` — a thin wrapper over a `fetch` that sets `X-Pd-App: ` and goes through Pipedream proxy. Returns `{status, headers, body_text}`. Same proxy `exec_code` uses. - `web_search(query)` — existing. - `fetch_docs(url)` — existing (Firecrawl-backed). -- `read_mirror(sync_id)` — reads `integrations//struct.json` via R2. +- `read_mirror(sync_id, path)` — reads one file from a sync's mirror prefix. `path` is relative (e.g. `tree.json`, `threads/by-label/INBOX.json`). Returns the body as a string or `null` if missing. +- `list_mirror(sync_id, prefix?)` — lists keys under the sync's mirror prefix (further filtered by an optional sub-prefix). Used during verify (`Phase 7`) to confirm the layout the agent wrote actually exists, and during debugging. - `query_sync(sync_id, sql)` — SELECT-only SQL against the facet's own SQLite. Used to read `raw_events` during the probe loop, inspect `agent_dedup` state during debugging, etc. Routed via supervisor RPC into the facet's SQL surface. - `get_sync_errors(sync_id, limit?)` — paginated rows from `sync_errors`, newest first, by `(sync_id, occurred_at DESC)`. Returns `{occurred_at, error_message, stack_trace, payload_preview, handler_version}` per row. Used by the sync agent during the probe loop (verify phase-2 signature checks) and by `update_sync` debugging. **Write tools (all of these mutate the kernel DO or R2):** - `write_skill({slug, body})` — writes `skills/integrations/.md`. Validates frontmatter. -- `create_sync_engine({integration, account_id, strategy, handler_js, schema_ddl, schema_doc, skill_path, mirror_path, poll_interval_sec?, renewal_interval_sec?})` — inserts the row in status='discovering', compiles the bundle via Worker Loader (rolls back on compile error), boots the facet (which runs `schema_ddl` against the facet's SQLite), calls `handler.registerWebhook` if applicable, sets first alarm. On success flips status to 'active'. On any failure: status='errored', errorText populated, returns the error. **Does NOT call backfill** — backfill is the agent's next step via `run_in_sync` (Phase 5.5). +- `create_sync_engine({integration, account_id, strategy, handler_js, schema_ddl, schema_doc, skill_path, mirror_prefix?, poll_interval_sec?, renewal_interval_sec?})` — inserts the row in status='discovering', compiles the bundle via Worker Loader (rolls back on compile error), boots the facet (which runs `schema_ddl` against the facet's SQLite), calls `handler.registerWebhook` if applicable, sets first alarm. `mirror_prefix` defaults to `integrations//`. On success flips status to 'active'. On any failure: status='errored', errorText populated, returns the error. **Does NOT call backfill** — backfill is the agent's next step via `run_in_sync` (Phase 5.5). - `update_sync({sync_id, handler_js?, schema_ddl?, migration_ddl?, schema_doc?, strategy?, poll_interval_sec?, renewal_interval_sec?})` — atomic update. Bumps `version`. If `handler_js` changes, recompiles. If `schema_ddl` changes, runs `migration_ddl` against the existing facet SQLite BEFORE swapping (sauna pattern; agent provides the migration as `ALTER TABLE` / data-move statements). On migration failure, rolls back the whole update. If strategy changes from `poll` → `webhook_*`, registers webhook; vice versa unregisters. -- `run_in_sync({sync_id, code})` — one-shot execution of agent-authored ES-module JS against a sync's facet (sauna pattern: separate Worker Loader bundle per call, not the persistent handler bundle). Code must export `async function runInSync(facet, fetch)` and return whatever the agent wants surfaced. `facet` exposes `__query(sql)` (read-only) and `__sqlExec(sql, ...bindings)` (writes/DDL) via the FacetBridge RpcTarget (raw DO stubs can't cross Worker Loader isolate boundaries — see sauna's run-in-sync.ts comment), plus `__readMirror()` / `__writeMirror(json)` for R2. `fetch` is Pipedream-proxied with the same `X-Pd-App` requirement as the handler. **This is the universal write-side channel for one-shot work**: initial backfill (Phase 5.5), diagnostics (read-only state inspection), data repair after a buggy handler version, and probe-loop triggers (firing a test event upstream so a webhook fires). Output capped at 64KB serialized JSON; logs from `console.log`/`info`/`warn`/`error` captured separately. Approval-gated. -- `delete_sync({sync_id})` — destructive. Calls `handler.unregisterWebhook` if applicable, deletes `skills/integrations/.md` from R2, deletes `integrations//struct.json`, calls `ctx.facets.delete(sync.id)` (which drops the facet's SQLite — `raw_events`, `agent_dedup`, everything goes), flips sync_engine row to status='deleted' (kept for audit). For Branch-B (api-only) syncs there's no sync_engine row to flip — `delete_sync` just removes the skill. +- `run_in_sync({sync_id, code})` — one-shot execution of agent-authored ES-module JS against a sync's facet (sauna pattern: separate Worker Loader bundle per call, not the persistent handler bundle). Code must export `async function runInSync(facet, fetch)` and return whatever the agent wants surfaced. `facet` exposes `__query(sql)` (read-only) and `__sqlExec(sql, ...bindings)` (writes/DDL) on the facet's SQLite, plus `__read(path)`, `__write(path, body)`, `__list(prefix?)`, `__delete(path)` on the R2 mirror (paths relative to mirror_prefix). All of this goes through the FacetBridge RpcTarget — raw DO stubs can't cross Worker Loader isolate boundaries (see sauna's run-in-sync.ts comment). `fetch` is Pipedream-proxied with the same `X-Pd-App` requirement as the handler. **This is the universal write-side channel for one-shot work**: initial backfill (Phase 5.5), diagnostics (read-only state inspection), data repair after a buggy handler version, and probe-loop triggers (firing a test event upstream so a webhook fires). Output capped at 64KB serialized JSON; logs from `console.log`/`info`/`warn`/`error` captured separately. Approval-gated. +- `delete_sync({sync_id})` — destructive. Calls `handler.unregisterWebhook` if applicable, deletes `skills/integrations/.md` from R2, **list-and-deletes every key under the sync's `mirror_prefix`** (R2 batch delete), calls `ctx.facets.delete(sync.id)` (which drops the facet's SQLite — `raw_events`, `agent_dedup`, everything goes), flips sync_engine row to status='deleted' (kept for audit). For Branch-B (api-only) syncs there's no sync_engine row to flip — `delete_sync` just removes the skill. - `manage_tasks` — existing. The sync agent uses it to track its own phases (good UX: "Phase 1: discovery 🔄 Phase 2: schema 🔄 ..."). Tools the sync agent does **not** get: @@ -813,7 +833,7 @@ The sync agent is **not** triggered by a slash command. Connecting an integratio - `/sync status` — lists every integration that has a `skills/integrations/.md`. For each: branch (sync-engine | api-only), and for sync-engine integrations also status (active / errored), strategy, last_synced_at, error_text if any. No LLM call. - `/sync rerun ` — re-publishes the connection event to the `integration-sync` queue, bypassing the idempotency check (forces re-run even if `sync_engine` row exists in status='active'). Used when (a) a previous run errored, (b) user reorganized their data and wants a fresh discovery, (c) handler bug → resync needed. -- `/sync delete ` — calls the `delete_sync` write tool directly (no agent involvement); confirms with the user first. Unregisters webhook upstream, drops mirror json, drops skill, calls `ctx.facets.delete(sync.id)`, flips row to status='deleted'. +- `/sync delete ` — calls the `delete_sync` write tool directly (no agent involvement); confirms with the user first. Unregisters webhook upstream, deletes every key under the sync's `mirror_prefix`, drops skill, calls `ctx.facets.delete(sync.id)`, flips row to status='deleted'. Implementation: same pattern as existing slash commands (under `apps/cli/src/commands/sync.ts` for the CLI side; server-side dispatch in the kernel). @@ -862,8 +882,8 @@ SyncAgent: create_sync_engine { strategy: "webhook_channel", handler_js: "", schema_doc: "", - skill_path: "skills/integrations/dropbox.md", - mirror_path: "integrations/dropbox/struct.json" + skill_path: "skills/integrations/dropbox.md" + // mirror_prefix omitted → defaults to "integrations/dropbox/" } Kernel: INSERT sync_engine row, status='discovering' Kernel: getOrCreateFacet → ctx.facets.get(sync.id, () => { ...buildHandlerBundle → getDurableObjectClass("Handler") }). Wrapper constructor runs. @@ -899,16 +919,17 @@ SyncAgent: run_in_sync { entries.push(...res.entries); cursor = res.cursor; } - await facet.__writeMirror({ _cursor: cursor, entries }); + await facet.__write('tree.json', JSON.stringify({ _cursor: cursor, _updated_at: new Date().toISOString(), entries })); return { entries: entries.length }; }" } Kernel: load exec bundle via env.LOADER.load(buildExecWorkerCode({ code, ... })) — separate from the handler bundle → calls FacetBridge.runInSync(facet, fetch) → exec returns { ok: true, result: { entries: 1247 }, logs: [...] } -SyncAgent: (sees {entries: 1247}; mirror is now populated; cursor is in _cursor) +SyncAgent: (sees {entries: 1247}; tree.json is now populated) SyncAgent: manage_tasks update "Sync dropbox: Verify" -SyncAgent: read_mirror sync_id → confirms shape +SyncAgent: list_mirror sync_id → ["tree.json"] +SyncAgent: read_mirror sync_id, "tree.json" → confirms shape SyncAgent: get_sync_errors sync_id → empty SyncAgent: manage_tasks complete "Sync dropbox: Done" ↓ @@ -927,16 +948,17 @@ Kernel: const facet = getOrCreateFacet(this, row) // ctx.facets.get Kernel: await facet.verifySignature(body, headers, row.webhook_secret) → true Kernel: await facet.handleWebhook(body, headers) ↓ (inside facet wrapper → user handler) - const mirror = await ctx.readMirror() // MirrorIO RPC to supervisor + const tree = JSON.parse(await ctx.mirror.read("tree.json")) // MirrorFS RPC to supervisor const r = await ctx.fetch("https://api.dropboxapi.com/2/files/list_folder/continue", { method: "POST", headers: { "X-Pd-App": "dropbox", "Content-Type": "application/json" }, - body: JSON.stringify({ cursor: mirror._cursor }) - }) // HttpGateway RPC to supervisor → Pipedream → Dropbox + body: JSON.stringify({ cursor: tree._cursor }) + }) // HttpGateway RPC to supervisor → Pipedream → Dropbox const data = await r.json() - // apply data.entries (added/modified) and data.entries with .tag='deleted' to mirror.entries - // update mirror._cursor = data.cursor - await ctx.writeMirror(mirror) // MirrorIO RPC to supervisor + // apply data.entries (added/modified) and data.entries with .tag='deleted' to tree.entries + tree._cursor = data.cursor + tree._updated_at = new Date().toISOString() + await ctx.mirror.write("tree.json", JSON.stringify(tree)) // MirrorFS RPC to supervisor return new Response("ok", { status: 200 }) ↓ Kernel: UPDATE sync_engine SET last_synced_at @@ -952,7 +974,7 @@ Main agent's turn begins. Context block: includes dropbox. System prompt teaches "check skills/integrations/.md". Skill auto-load: skills/integrations/dropbox.md is in scope. Main agent decides: this is a dropbox structure question, mirror has the answer. -Main agent: read_file integrations/dropbox/struct.json (returns the full json — file tree, no contents) +Main agent: read_file integrations/dropbox/tree.json (returns the full json — file tree, no contents) Main agent: (inline grep for path starts with /Personal/Tax/) Main agent: Replies with the list of files + brief summary. ``` @@ -969,7 +991,7 @@ Skill auto-load: skills/integrations/dropbox.md is in scope. Skill says: headers: { 'X-Pd-App': 'dropbox', 'Dropbox-API-Arg': JSON.stringify({path: '/path/to/file'}) } })" -Main agent: read_file integrations/dropbox/struct.json +Main agent: read_file integrations/dropbox/tree.json Main agent: (inline grep for /Personal/Tax/ AND modified > 2025-01-01) → 4 matching entries, with paths Main agent: (decides to summarize — needs contents) @@ -1039,20 +1061,20 @@ exec_code returns success. - `packages/models/src/schema/sync-engine.ts` — `sync_engine` + `sync_errors` drizzle tables (full schema as in §3.3) - `apps/kernel/src/sync/facet.ts` — `getOrCreateFacet(kernel, sync)`, `buildHandlerBundle(handlerJs)`, the facet wrapper entrypoint factory (`Handler` class with one RPC per export + `alarm()` dispatcher) - `apps/kernel/src/sync/http-gateway.ts` — `HttpGateway` WorkerEntrypoint (Pipedream-proxied fetch, scoped by `props.integration` + `props.accountId`) -- `apps/kernel/src/sync/mirror-io.ts` — `MirrorIO` WorkerEntrypoint (R2 read/write scoped by `props.mirrorPath`) +- `apps/kernel/src/sync/mirror-fs.ts` — `MirrorFS` WorkerEntrypoint exposing `read(path)`, `write(path, body)`, `list(prefix?)`, `delete(path)` over R2 scoped by `props.mirrorPrefix` (prepends the prefix, rejects `..` and leading `/`) - `apps/kernel/src/sync/dispatch.ts` — `dispatchWebhook` body (looks up sync row, calls `getOrCreateFacet`, RPCs verifySignature + handleWebhook) - `apps/cli/src/commands/sync.ts` — `/sync status` and `/sync delete` (no `/sync ` trigger; queue is the trigger) - `__tests__/fixtures/dropbox-handler.js` — hand-written smoke fixture (verifySignature, handleWebhook with cursor + list_folder/continue, registerWebhook, unregisterWebhook). No backfill export — that runs via the exec runner. -- `apps/kernel/src/sync/exec-runner.ts` — `buildExecWorkerCode({code, syncId, ...})` + `FacetBridge` RpcTarget (mirrors `__query`, `__sqlExec`, `__readMirror`, `__writeMirror` so agent code can talk to the facet across the Worker Loader isolate boundary — sauna's `src/execrunner/*`) +- `apps/kernel/src/sync/exec-runner.ts` — `buildExecWorkerCode({code, syncId, ...})` + `FacetBridge` RpcTarget mirroring `__query`, `__sqlExec` (facet SQLite) and `__read`, `__write`, `__list`, `__delete` (MirrorFS) so agent code can talk to the facet's storage layers across the Worker Loader isolate boundary — sauna's `src/execrunner/*` - `__tests__/fixtures/dropbox-backfill-exec.js` — hand-written smoke fixture for the backfill path (`runInSync(facet, fetch)` that walks list_folder, writes mirror) **Files modified:** -- `apps/kernel/src/index.ts` — add `POST /webhook/:integration/:account_id` route; export `HttpGateway` and `MirrorIO` WorkerEntrypoints alongside existing PipedreamProxy etc. +- `apps/kernel/src/index.ts` — add `POST /webhook/:integration/:account_id` route; export `HttpGateway` and `MirrorFS` WorkerEntrypoints alongside existing PipedreamProxy etc. - `apps/kernel/src/kernel.ts` — add `dispatchWebhook` RPC method, add supervisor-side helper for `sync_errors` row inserts (called via RPC by the facet wrapper on failure) - `apps/kernel/src/agent/system-prompt.ts` — add the convention sentence: "For any integration in ``, `skills/integrations/.md` (if present) contains usage guidance — read it before invoking that integration." No new injected block. -- `wrangler.jsonc` — declare `HttpGateway` and `MirrorIO` as named WorkerEntrypoints if not auto-discovered +- `wrangler.jsonc` — declare `HttpGateway` and `MirrorFS` as named WorkerEntrypoints if not auto-discovered -**Smoke:** install fixture handler via a dev-only RPC that simulates `create_sync_engine` → fixture's `registerWebhook` runs against real Dropbox (Pipedream-proxied) → change a file in Dropbox → empty ping arrives → facet handles → `integrations/dropbox/struct.json` updates within 30s. No agent and no queue consumer involved. +**Smoke:** install fixture handler via a dev-only RPC that simulates `create_sync_engine` → fixture's `registerWebhook` runs against real Dropbox (Pipedream-proxied) → change a file in Dropbox → empty ping arrives → facet handles → `integrations/dropbox/tree.json` updates within 30s. No agent and no queue consumer involved. ### PR-F2 — Queue trigger + sync agent + Dropbox e2e **Files created:** @@ -1069,7 +1091,7 @@ exec_code returns success. - `apps/cli/src/commands/sync.ts` — add `/sync rerun ` — POSTs to `/integrations/connected` with `force: true` bit that bypasses the idempotency check inside `runSyncAgent` - `wrangler.jsonc` — declare `integration-sync` queue (producer + consumer bindings; DLQ; max retries) -**Smoke:** complete `/connect dropbox` in CLI (or simulate by direct POST to `/integrations/connected`) → queue receives event → consumer fires → sync agent run begins in system thread → triage decides Branch A → ~5 min later, `skills/integrations/dropbox.md` (sync-engine flavor) and `integrations/dropbox/struct.json` exist; main agent (in any thread) sees `dropbox` in `` and reads the skill; ask "where are my tax filings?" → answer correct; main agent uses `exec_code` to upload a file → webhook fires → mirror reflects within 30s; `/sync delete dropbox` cleans up upstream + R2. Then ALSO smoke Branch B: connect MailChimp → queue → triage decides Branch B → ~30s later, `skills/integrations/mailchimp.md` (api-only flavor) exists, no sync_engine row, no struct.json. +**Smoke:** complete `/connect dropbox` in CLI (or simulate by direct POST to `/integrations/connected`) → queue receives event → consumer fires → sync agent run begins in system thread → triage decides Branch A → ~5 min later, `skills/integrations/dropbox.md` (sync-engine flavor) and the agent-chosen file(s) under `integrations/dropbox/` exist; main agent (in any thread) sees `dropbox` in `` and reads the skill; ask "where are my tax filings?" → answer correct; main agent uses `exec_code` to upload a file → webhook fires → mirror reflects within 30s; `/sync delete dropbox` cleans up upstream + every R2 key under the prefix. Then ALSO smoke Branch B: connect MailChimp → queue → triage decides Branch B → ~30s later, `skills/integrations/mailchimp.md` (api-only flavor) exists, no sync_engine row, no files under `integrations/mailchimp/`. ### PR-F3 — Polling fallback + research wave + additional integrations **Files created:** @@ -1114,20 +1136,20 @@ exec_code returns success. 1. `pnpm dev` → CLI boots, no regressions in existing flows. 2. Drizzle migration applies cleanly. 3. `/sync status` from clean state prints "no active syncs." -4. Dev-only RPC installs the fixture dropbox handler into `sync_engine` with status='active', strategy='webhook_channel', mirror_path='integrations/dropbox-fixture/struct.json'. +4. Dev-only RPC installs the fixture dropbox handler into `sync_engine` with status='active', strategy='webhook_channel', mirror_prefix='integrations/dropbox-fixture/'. 5. Fixture's `registerWebhook` runs successfully against real Dropbox API (Pipedream-proxied, against a test account). 6. Webhook URL `https:///webhook/dropbox/` returns 200 to a signed Dropbox POST. -7. Touch a file in Dropbox → empty ping arrives → handler reads cursor, calls list_folder/continue, applies delta → `integrations/dropbox-fixture/struct.json` reflects the change within 30s. +7. Touch a file in Dropbox → empty ping arrives → handler reads cursor from `tree.json`, calls list_folder/continue, applies delta → `integrations/dropbox-fixture/tree.json` reflects the change within 30s. 8. Bad-signature request returns 401, `sync_errors` row recorded with `error_message="signature verification rejected"`. 9. Main agent's system prompt has the convention sentence pointing at `skills/integrations/.md`; with the fixture installed, `/sync status` confirms the fixture row exists. -10. `/sync delete dropbox` removes the row, drops the mirror json, fixture's unregisterWebhook is called. +10. `/sync delete dropbox` removes the row, list-and-deletes every key under `integrations/dropbox-fixture/`, fixture's unregisterWebhook is called. ### PR-F2 smoke 1. From zero state (no fixture, no sync_engine row), simulate the connection event: `curl -X POST http://localhost:8787/integrations/connected -d '{"integration":"dropbox","account_id":"acc_xyz"}'`. 2. Queue receives the message; consumer fires; sync agent run begins (visible as a new run row in the `system` thread, queryable via DB). 3. Sync agent completes within ~5–10 min (target; cost ceiling enforced). 4. `skills/integrations/dropbox.md` exists in R2 and accurately describes the test account's folder structure in prose. -5. `integrations/dropbox/struct.json` exists, has the expected shape, includes top-level folders. +5. `integrations/dropbox/tree.json` (or whatever filename the agent chose — likely tree.json given §3.4 example) exists, has the expected shape, includes top-level folders. `list_mirror` over the prefix returns the files the skill's "Mirror layout" section claims will be there. 6. Main agent (next turn, in any user thread) sees `dropbox` in `` and reads `skills/integrations/dropbox.md` when relevant. 7. Ask main agent: "where are my tax filings?" → it reads the skill, greps the mirror, answers correctly with paths. 8. Ask main agent: "upload a one-pager for AcmeCorp under the right portfolio folder" → it picks the path from the skill, runs `exec_code`, file lands in Dropbox. @@ -1154,8 +1176,9 @@ exec_code returns success. | 3 | No `api_hosts` allowlist | Pipedream proxy is the egress gate; redundant in agent-os. | | 4 | No `ctx.agent` in handler runtime | Per user; inline JS handlers are enough; runtime LLM reasoning out of scope. | | 5 | Agent authors json shape | Sauna parity; "machine builds machine"; schema_doc carries the human-readable description. | -| 6 | Cursor lives inside the mirror json (reserved `_cursor` field) by default; agent can override into facet SQLite via `cursors` table if multi-cursor | Default keeps cursor atomic with mirror writes. For multi-resource integrations (Drive watch channels per folder, etc.), agent can declare a `cursors` table in `schema_ddl` and store it in facet SQLite instead. | -| 6b | Two schemas authored per Branch-A sync: `schema_doc` (prose, describes R2 mirror json) + `schema_ddl` (SQL, defines facet SQLite tables) | Two distinct data layers serve different purposes. Mirror is structural index for main agent (R2 json; grep-friendly). Facet SQLite is handler-internal state (dedup keys, raw_events for probe loop, multi-cursor state). The agent designs both at create_sync_engine time; supervisor runs DDL at facet bootstrap. | +| 6 | Cursor location is the agent's call: embedded in a mirror file (e.g. `tree.json#_cursor` or `cursors.json`), or in facet SQLite | No default forced shape. For low-churn cursors atomic with a small mirror write, the mirror is simplest. For high-churn or multi-resource cursors (Drive watch channels per folder), facet SQLite avoids R2 writes on every webhook. Agent picks at design time. | +| 6b | Two schemas authored per Branch-A sync: `schema_doc` (prose, describes the R2 mirror layout — which files exist, what shape each has) + `schema_ddl` (SQL, defines facet SQLite tables) | Two distinct data layers serve different purposes. Mirror is structural index for main agent (R2 KV space at `integrations//`, grep-friendly per file). Facet SQLite is handler-internal state (dedup keys, raw_events for probe loop, multi-cursor state). The agent designs both at create_sync_engine time; supervisor runs DDL at facet bootstrap. | +| 6c | Mirror is a prefix-scoped R2 key-value space, not a single file | Single-file forces whole-tree rewrites on every delta (expensive for high-write integrations like Gmail) and makes large mirrors hostile to grep-friendly `read_file`. Prefix layout lets the agent split by useful axis (label, repo, base, etc.); main agent finds the right file via the skill md's "Mirror layout" section. MirrorFS WorkerEntrypoint enforces the prefix scope; facet can't escape. | | 7 | Sync agent is NOT a tool — it runs in a Cloudflare Queue consumer triggered by connection event | Decouples connection detection from agent execution. Sync agent runs are minutes-long and burn tokens — running them inline on the connection-detection path would block other work and blow Worker timeouts. Queue gives retries, DLQ, back-pressure for free. | | 7b | Sync agent runs on EVERY connection; triages first | 3000+ Pipedream integrations; most are API-wrappers with nothing to sync. Hand-curated allowlist would rot fast. Agent's first phase decides "Branch A (full sync engine)" vs "Branch B (skill-only)". Branch B costs ~30s of one cheap run; small enough to absorb. | | 7c | NO `` injected block; rely on existing `` + skill convention | Pre-injecting metadata for N connected integrations is context bloat at scale. The skill md file IS the description — the agent reads it lazily when the question touches that integration. One system-prompt sentence teaches the convention. | From 0b1dc065b0cfcdd797061bf1f40ef973ec23c5de Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 09:57:03 +0530 Subject: [PATCH 04/13] docs(sync-agent): primitives stay neutral; system prompt teaches state placement MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit User feedback: spec was over-prescribing where cursors live (mirror-first framing). Cleaner pattern: agent declares `webhook_config` table in schema_ddl (channel_id, expires_at, history_id) and UPDATEs that row on every webhook — faster and simpler than read-parse-mutate-write on a JSON file. Bigger principle: primitives don't teach behavior, prompts do. - New §1.2: 'Primitives stay neutral; the system prompt teaches behavior'. Codifies that placement (cursor, layout, schema, surface choice) is always the agent's call; the runtime imposes no defaults. - §3.4: rewrote cursor-placement paragraph to list three peer options (mirror-embedded / sidecar file / SQLite row); SQLite is the preferred default for high-churn or multi-field webhook state. - §3.4 Gmail example: shows `webhook_config` SQL table as the cursor home (single-row UPDATE per webhook), keeps mirror for grep-friendly index; notes the cursors.json alternative as equally valid for lower-volume integrations. - §3.6 'Why this is enough for Gmail': updated to SQL cursor path with R2/SQL op counts. - §3.10.3 Phase 2 schema_ddl example: replaces generic `cursors` example with both `webhook_config` (Gmail shape) and `cursors` (Drive shape). - §3.10.3 Phase 4: new paragraph teaching the SQL-vs-mirror placement rule of thumb (main agent reads it → mirror; only handler reads it → SQL; when in doubt → SQL). - §10 decisions: row 6 reframed (SQLite is the preferred default for webhook state); new row 13 codifying the neutral-primitives principle. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../specs/2026-05-17-sync-agent-design.md | 83 ++++++++++++++++--- 1 file changed, 70 insertions(+), 13 deletions(-) diff --git a/docs/superpowers/specs/2026-05-17-sync-agent-design.md b/docs/superpowers/specs/2026-05-17-sync-agent-design.md index a598fe7..98d21c0 100644 --- a/docs/superpowers/specs/2026-05-17-sync-agent-design.md +++ b/docs/superpowers/specs/2026-05-17-sync-agent-design.md @@ -50,6 +50,18 @@ Per-integration heuristics the agent should apply: This principle gets baked into the sync agent's system prompt as a top-level directive (§3.10.3, Phase 2 — Schema design). If the agent starts to mirror content, it has misread the task. +### 1.2 Primitives stay neutral; the system prompt teaches behavior + +A meta-principle that constrains the rest of this spec: the supervisor, the schema, the facet wrapper, and the tool surface should give the sync agent unopinionated primitives — read/write/list mirror, query/exec SQL, fetch via Pipedream — and stay quiet about *how* to use them. Best practices live in the sync agent's system prompt (§3.10.3), not in the data model or the runtime. + +Concretely: +- **Cursor placement** is the agent's call (mirror field, sidecar file, or SQLite row — §3.4). The supervisor never reads cursors. +- **Mirror layout** is the agent's call (one file, many files, nested prefixes — §3.4). MirrorFS doesn't introspect. +- **State partitioning between mirror and SQLite** is the agent's call. Both surfaces are always wired. +- **DDL shape** is the agent's call. The supervisor blindly runs whatever `schema_ddl` it's handed. + +The system prompt teaches *when* the SQLite path is better (high-churn webhook state, multi-field config rows like `{channel_id, expires_at, history_id}` updated atomically per delivery), *when* the mirror is better (anything the main agent needs to grep), and *when* both are right (mirror as the index, SQLite as the operational state). LLMs apply judgment to primitives better than they apply judgment to over-prescribed runtimes; this principle keeps the runtime out of the agent's way. + This is sauna-shaped, with two deliberate departures and one inherited primitive: - **Inherited from sauna: per-sync Cloudflare DO Facets.** Each sync runs inside its own DO facet (`ctx.facets.get(sync.id, ...)`). The facet has isolated SQLite (handler scratch space — dedup tables, ephemeral state), receives capabilities from the Kernel DO supervisor via RPC, and self-schedules its own renewal/poll alarms. Lifecycle primitives (`ctx.facets.abort()` on update, `ctx.facets.delete()` on delete) map cleanly to `update_sync` / `delete_sync`. Facets are not separately billed. @@ -165,7 +177,13 @@ Only inserted when something throws: DESC index on `(sync_id, occurred_at)` supports `SELECT ... ORDER BY occurred_at DESC LIMIT N` efficiently — that's the `get_sync_errors` read pattern. Cascade-deletes when the parent sync row is removed. -Cursors live in the mirror — but the agent decides *where* in the mirror. Could be a reserved `_cursor` field in the single tree file, a sidecar `cursors.json`, a per-resource map (e.g. `cursors/.json`), or a row in the facet's SQLite if the cursor is high-churn. Whatever fits the integration. The supervisor doesn't care. +**Cursors and webhook state live wherever the agent decides.** The supervisor doesn't care. Three first-class options the system prompt will teach (§3.10.3); the agent picks per integration: + +- **Embedded in a mirror file** (`tree.json#_cursor`). Best when the cursor is low-churn AND atomic with the file it gates — Dropbox's single tree, Airtable's schema-of-schemas. +- **Sidecar mirror file** (`cursors.json`). Best when cursor churn is medium AND the main index is large enough that you don't want to rewrite it on every cursor bump. +- **Facet SQLite row** (`webhook_config` table: `channel_id`, `expires_at`, `history_id`, `last_seen_at`, ...). Best when state churns per-webhook OR there's more than just a cursor (registration ids, expiry, per-resource pointers). A single `UPDATE webhook_config SET history_id=?, last_seen_at=?` on every webhook beats `read JSON → parse → mutate → stringify → R2 write` — by a lot. Gmail's per-label history pointers and Drive's per-folder channel state are the canonical fits. + +The handler/exec code reaches SQL via `ctx.sql` (handler) or `facet.__query`/`facet.__sqlExec` (run_in_sync); reaches the mirror via `ctx.mirror.*` (handler) or `facet.__read`/`facet.__write` (run_in_sync). Both surfaces are always available — the agent picks what to use, the supervisor doesn't impose. ### 3.4 The mirror — a prefix-scoped key-value space in R2 @@ -203,21 +221,40 @@ integrations/dropbox/ `schema_doc`: > Single `tree.json` at the prefix root. `entries` is a flat array of `{id, path, name, type, size, modified, mime}`. `_cursor` is Dropbox's list_folder/continue cursor. File contents NOT mirrored; download via `https://content.dropboxapi.com/2/files/download` with `X-Pd-App: dropbox` from `exec_code`. -**Gmail (high-volume, multiple useful slices) — multi-file pays off.** +**Gmail (high-volume, multiple useful slices) — multi-file mirror + SQLite cursor pays off.** + +The mirror holds what the main agent needs to grep; the high-churn operational state lives in facet SQLite where a single `UPDATE` per webhook beats read-parse-mutate-write on a JSON file: ``` integrations/gmail/ ├── labels.json # { labels: [{id, name, type, ...}] } — small, queried often -├── cursors.json # { historyId, perLabelLastSeen } — separated so handler can rewrite without touching labels └── threads/ └── by-label/ - ├── INBOX.json # { threads: [{id, snippet, from, subject, date, hasAttachments}, ...] } — last N threads per label + ├── INBOX.json # { threads: [{id, snippet, from, subject, date, hasAttachments}, ...] } ├── SENT.json └── Label_2742.json ``` +```sql +-- facet SQLite (schema_ddl) +CREATE TABLE IF NOT EXISTS webhook_config ( + channel_id TEXT PRIMARY KEY, -- Pub/Sub watch resource id + topic_name TEXT NOT NULL, + history_id TEXT NOT NULL, -- Gmail's cursor — UPDATEd on every webhook + expires_at INTEGER NOT NULL, -- ms epoch; renewWebhook uses this + last_seen_at INTEGER NOT NULL +); + +CREATE TABLE IF NOT EXISTS per_label_cursor ( + label_id TEXT PRIMARY KEY, + last_history TEXT NOT NULL +); +``` + `schema_doc`: -> `labels.json` is the full label list. `cursors.json` tracks Gmail History API `historyId` plus per-label last-seen markers. `threads/by-label/.json` holds the most recent 200 thread headers per label (id, snippet, from, subject, date, hasAttachments) — full bodies fetched on demand via Gmail API. Main agent finds threads by reading the right label file directly (e.g. `read_file integrations/gmail/threads/by-label/INBOX.json`); doesn't need to grep a global thread list. +> Mirror: `labels.json` (full label list) and `threads/by-label/.json` (most recent 200 thread headers per label — id, snippet, from, subject, date, hasAttachments). Full bodies fetched on demand via Gmail API. Cursor state (Gmail History `historyId`, per-label last-seen, watch-channel registration + expiry) lives in facet SQLite tables `webhook_config` and `per_label_cursor` — single-row updates per webhook, no R2 traffic on the hot path. Main agent finds threads by reading the right label file directly (e.g. `read_file integrations/gmail/threads/by-label/INBOX.json`). + +A perfectly fine alternative: keep cursor state in `integrations/gmail/cursors.json` instead of SQLite. The trade is one R2 read+write per webhook vs one SQL UPDATE. For Gmail's webhook volume the SQL route wins; for an integration with one webhook a day it doesn't matter. **The agent decides.** **Airtable (schema-of-schemas) — single file again; schema is small.** @@ -492,7 +529,7 @@ That's it. No supervisor SQL access (only facet's own sqlite). No sub-agent spaw **Why this is enough for Dropbox:** the handler `mirror.read("tree.json")` (parses it to get `_cursor`), calls `https://api.dropboxapi.com/2/files/list_folder/continue` with the cursor, receives a list of file changes, applies them to `entries`, updates `_cursor`, `mirror.write("tree.json", JSON.stringify(next))`. ~30 lines of JS. -**Why this is enough for Gmail:** handler receives a webhook for label X, `mirror.read("cursors.json")` to get the per-label cursor, fetches the history slice via the Gmail API, `mirror.read("threads/by-label/.json")`, applies the delta to that slice only, writes it back, updates `cursors.json`. The unchanged label files stay untouched. +**Why this is enough for Gmail:** handler receives a webhook for label X, does `SELECT history_id FROM webhook_config` via `ctx.sql` (single SQL read), fetches the history slice via the Gmail API, `mirror.read("threads/by-label/.json")`, applies the delta to that slice only, writes it back, runs `UPDATE webhook_config SET history_id = ?, last_seen_at = ?` (single SQL write). Two R2 ops per webhook (one read + one write of the one changed label file) and two SQL ops — vs three R2 ops if the cursor lived in `cursors.json`. The unchanged label files stay untouched either way. ### 3.7 Per-sync facet (Cloudflare DO Facets primitive) @@ -744,18 +781,30 @@ CREATE TABLE IF NOT EXISTS agent_dedup ( claimed_at INTEGER NOT NULL ); --- Optional: cursor in SQL if you'd rather not put it in a mirror file. --- For Dropbox: leave cursor in tree.json's _cursor field — simpler. --- For Gmail: cursors.json works fine (one file, low write churn). --- For high-churn multi-cursor integrations (Google Drive: one cursor per --- change-channel, updated on every webhook), facet SQLite is cleaner since --- the cursor write doesn't touch R2 on every single webhook: +-- Webhook/cursor state. Choosing SQL over a mirror file is usually correct +-- when state churns per delivery, when there's more than just a cursor +-- (registration ids, expiries, per-resource pointers), or when the upstream +-- has multiple resources to track. One SQL UPDATE per webhook beats +-- read-parse-mutate-write on a JSON file. Example for Gmail: +CREATE TABLE IF NOT EXISTS webhook_config ( + channel_id TEXT PRIMARY KEY, + topic_name TEXT NOT NULL, + history_id TEXT NOT NULL, + expires_at INTEGER NOT NULL, + last_seen_at INTEGER NOT NULL +); + +-- A simpler shape — one row per resource, one cursor each — works for +-- integrations like Drive (channel per folder): CREATE TABLE IF NOT EXISTS cursors ( resource_id TEXT PRIMARY KEY, cursor TEXT NOT NULL, updated_at INTEGER NOT NULL ); +-- For low-churn cursors atomic with a small mirror, skip SQL entirely and +-- embed the cursor in the mirror file (e.g. tree.json#_cursor for Dropbox). + -- For the probe-loop's phase-1 raw capture (Phase 6): CREATE TABLE IF NOT EXISTS raw_events ( id INTEGER PRIMARY KEY AUTOINCREMENT, @@ -782,6 +831,13 @@ Write `skills/integrations/.md` with frontmatter `type: sync-engine` and a Pick the strategy (`webhook_stable` / `webhook_channel` / `poll`) based on integration capabilities — research with `fetch_docs` if unsure. Write the handler module string. The handler's job is to keep the STRUCTURAL index current — on every webhook delivery it updates the index entries (file moved? issue retitled? record schema changed?), it does NOT pull down content. Required exports for each strategy as in §3.6. +**Where state lives is your call.** Two surfaces are always wired and you pick per piece of state: + +- **Facet SQLite** (`ctx.sql.exec(...)` / `ctx.sql.exec(...).toArray()`). Best for high-churn webhook state — registration ids, channel expiries, history pointers, per-resource cursors. A single-row `UPDATE webhook_config SET history_id=?, expires_at=?, last_seen_at=?` on every delivery beats `mirror.read → JSON.parse → mutate → JSON.stringify → mirror.write` by a wide margin once webhook volume is non-trivial. Also the right home for dedup keys, raw_events during the probe loop, and any bookkeeping the main agent shouldn't see. +- **R2 mirror** (`ctx.mirror.read/write/list/delete`). Best for anything the main agent needs to grep — the structural index itself, schema docs cached for fast lookup, layout that the skill md advertises. + +Rule of thumb: if the data is read by the main agent through `read_file`, it belongs in the mirror. If it's only read by your own handler code, prefer SQL — UPDATE is faster, atomic, and doesn't round-trip through R2. Cursors specifically: SQL when the integration delivers webhooks faster than once per minute or has more than one cursor to track; mirror-embedded when it's one low-churn cursor atomic with a small file (Dropbox). When in doubt, SQL — the cost of wrong-placement in the mirror direction is far higher than wrong-placement in the SQL direction. + #### Phase 5 — Install (Branch A) Call `create_sync_engine({integration, accountId, strategy, handlerJs, schemaDoc, skillPath, mirrorPrefix?, pollIntervalSec?, renewalIntervalSec?})`. `mirrorPrefix` defaults to `integrations//` — agent only passes a custom value for unusual cases (multi-account isolation, manual namespacing). The kernel: compiles via Worker Loader (rolls back on compile failure), calls `handler.registerWebhook` if applicable, sets the initial alarm if needed. Backfill is NOT part of this call. @@ -1176,7 +1232,7 @@ exec_code returns success. | 3 | No `api_hosts` allowlist | Pipedream proxy is the egress gate; redundant in agent-os. | | 4 | No `ctx.agent` in handler runtime | Per user; inline JS handlers are enough; runtime LLM reasoning out of scope. | | 5 | Agent authors json shape | Sauna parity; "machine builds machine"; schema_doc carries the human-readable description. | -| 6 | Cursor location is the agent's call: embedded in a mirror file (e.g. `tree.json#_cursor` or `cursors.json`), or in facet SQLite | No default forced shape. For low-churn cursors atomic with a small mirror write, the mirror is simplest. For high-churn or multi-resource cursors (Drive watch channels per folder), facet SQLite avoids R2 writes on every webhook. Agent picks at design time. | +| 6 | Webhook/cursor state placement is the agent's call: facet SQLite row (preferred default for high-churn or multi-field state — Gmail's `webhook_config: {channel_id, expires_at, history_id}`, Drive's per-channel rows), mirror-embedded (`tree.json#_cursor` for Dropbox's single low-churn cursor), or sidecar mirror file. **The supervisor never reads cursors** — placement is purely a handler-internal performance decision. Sync agent system prompt (§3.10.3 Phase 4) teaches the rule of thumb: if only the handler reads it, prefer SQL; if the main agent reads it, mirror. The runtime imposes no defaults. | | 6b | Two schemas authored per Branch-A sync: `schema_doc` (prose, describes the R2 mirror layout — which files exist, what shape each has) + `schema_ddl` (SQL, defines facet SQLite tables) | Two distinct data layers serve different purposes. Mirror is structural index for main agent (R2 KV space at `integrations//`, grep-friendly per file). Facet SQLite is handler-internal state (dedup keys, raw_events for probe loop, multi-cursor state). The agent designs both at create_sync_engine time; supervisor runs DDL at facet bootstrap. | | 6c | Mirror is a prefix-scoped R2 key-value space, not a single file | Single-file forces whole-tree rewrites on every delta (expensive for high-write integrations like Gmail) and makes large mirrors hostile to grep-friendly `read_file`. Prefix layout lets the agent split by useful axis (label, repo, base, etc.); main agent finds the right file via the skill md's "Mirror layout" section. MirrorFS WorkerEntrypoint enforces the prefix scope; facet can't escape. | | 7 | Sync agent is NOT a tool — it runs in a Cloudflare Queue consumer triggered by connection event | Decouples connection detection from agent execution. Sync agent runs are minutes-long and burn tokens — running them inline on the connection-detection path would block other work and blow Worker timeouts. Queue gives retries, DLQ, back-pressure for free. | @@ -1188,3 +1244,4 @@ exec_code returns success. | 10 | Dropbox is the v1 e2e smoke target despite being archetype B | Matches the user's VC-example demo. Channel + cursor + empty-ping primitives are needed in F1/F2 anyway since Dropbox uses them. | | 11 | 3-PR stack (not 4) | Foundation primitives + queue trigger + agent + dropbox e2e split as F1/F2; F3 covers the other archetypes (Linear stable, gcal channel, Notion poll-or-webhook-bus). | | 12 | Master spec + per-PR plans | One coherent architecture doc; per-PR plans land when each PR kicks off so they reflect the implementation reality. | +| 13 | **Primitives stay neutral; the system prompt teaches behavior** (§1.2) | The supervisor exposes unopinionated read/write/list/delete + sql.exec + fetch primitives. It never imposes where cursors go, how the mirror is shaped, what tables to declare, or which surface (mirror vs SQL) is "correct" for a given piece of state. Best practices live in the sync agent's system prompt where the LLM can apply judgment. Adding defaults to the runtime would prematurely lock in choices that vary per integration; adding them to the prompt keeps the runtime flexible AND gives us a single place to update guidance as we learn what works. | From 4a273877921ea89f31017b4d22c4147fea22cd3a Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 11:05:27 +0530 Subject: [PATCH 05/13] =?UTF-8?q?docs(sync-agent):=20drop=20thread/run/mes?= =?UTF-8?q?sage=20machinery=20=E2=80=94=20plain=20streamText=20in=20consum?= =?UTF-8?q?er?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit User correction: sync agent is just an ai-sdk streamText invocation living inside the queue consumer. No thread, no run row, no message persistence, no streaming UI, no tool_call audit. User never sees it. Side effects (skill md, sync_engine row, mirror files, registered webhook) ARE the output. Kernel methods exist purely as RPC backends for tools that need to touch the DO sqlite — no runSyncAgent RPC, no system thread, no runSubAgent extraction. - §1 goal: reframed flow — consumer calls runSyncAgent({env, integration, accountId, kernelStub}) directly (not a DO RPC method) - §3.10: rewrote — sync agent is a plain streamText invocation in the queue consumer; included pseudocode skeleton showing tools wired via closure on kernelStub; explained why no thread/run/audit needed and debuggability via wrangler tail - §3.10.2: queue ack semantics updated; debugging via wrangler tail + sync_errors instead of system-thread browsing - §3.10.4: tool surface clarified — plain tool() from "ai" (no wrappedTool); buildSyncAgentTools takes {env, integration, accountId, kernelStub}; manage_tasks removed (no per-thread tasks table to write to) - §4.1: end-to-end trace rewritten — Consumer invokes runSyncAgent directly; tool execute fns RPC into kernelStub; no Kernel DO orchestrates the agent run, only services the tool calls - §6 PR-F2: added sync-agent.ts (runSyncAgent free function); dropped run-sub-agent.ts extraction + spawn-sub-agent refactor + system-thread bootstrap from kernel.ts; renamed Kernel methods to the tool-backend RPCs that actually need to exist (skillExists, writeSkill, createSyncEngine, runInSync, etc.) - §10 decisions: rewrote rows 8 & 9 — no thread/run machinery; no runSubAgent extraction. Rationale: spawn_sub_agent's persistence machinery exists because sub-agent output is visible to the user; sync agent isn't, so all of that infrastructure is irrelevant. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../specs/2026-05-17-sync-agent-design.md | 155 ++++++++++-------- 1 file changed, 90 insertions(+), 65 deletions(-) diff --git a/docs/superpowers/specs/2026-05-17-sync-agent-design.md b/docs/superpowers/specs/2026-05-17-sync-agent-design.md index 98d21c0..c7ef1a6 100644 --- a/docs/superpowers/specs/2026-05-17-sync-agent-design.md +++ b/docs/superpowers/specs/2026-05-17-sync-agent-design.md @@ -9,7 +9,7 @@ ## 1. Goal -When a user connects a third-party integration via the existing `/connect` flow, the connection event flows: client detection → kernel API → **`integration-sync` Cloudflare Queue** → queue consumer → `kernel.runSyncAgent(integration, account_id)`. That RPC spins up a **sync agent run** (sauna-style sub-agent, no thread binding) which: +When a user connects a third-party integration via the existing `/connect` flow, the connection event flows: client detection → kernel API → **`integration-sync` Cloudflare Queue** → queue consumer → `runSyncAgent({env, integration, accountId, kernelStub})`. That call is a plain AI-SDK `streamText` invocation running inside the consumer — no thread, no run row, no message persistence, no streaming UI. The agent's side effects (skill md, sync_engine row, mirror files, registered webhook) ARE its output. The user never sees the transcript and doesn't know it ran. The agent: 1. **Triages the integration.** Pipedream exposes 3,000+ integrations and most are API-wrappers with no syncable surface (LinkedIn, WhatsApp, MailChimp, Twilio, OpenAI, Sendgrid, Stripe-write, ...). Only a small subset have a navigable structure worth mirroring (file systems: Dropbox/Drive/OneDrive; databases: Airtable/Notion-DB; ticket trackers: Linear/GitHub/Jira; calendars: Google Calendar; wikis: Notion-pages/Confluence; messaging: Slack). The sync agent's first job is deciding which bucket this integration falls into. @@ -646,48 +646,73 @@ So the supervisor (Kernel DO) owns the single alarm. Facets receive renewal/poll ### 3.10 The sync agent -The sync agent is **not** a tool exposed to any LLM. It runs as a queue-consumer-triggered sub-agent inside the Kernel DO. The user never types `/sync ` to launch it — connecting the integration in `/connect` is the trigger. +The sync agent is **not** a sub-agent of the main agent, **not** a tool, and has **no thread, no run row, no message persistence, no streaming**. It is a plain `streamText` invocation from the AI SDK that runs inside the Cloudflare Queue consumer. The user never sees its transcript and has no idea it's running. The only output anyone observes is the side effects: a new `skills/integrations/.md`, optionally a `sync_engine` row + facet + webhook + mirror files. **Trigger pipeline:** ``` User completes /connect in browser ↓ -Kernel observes new Pipedream account (via existing diffAccounts + /connect refresh poll OR direct client signal — see §3.10.1) +CLI POSTs /integrations/connected { integration, account_id } to the Kernel worker +(or Kernel detects via existing diffAccounts polling — see §3.10.1) ↓ -Kernel publishes to `integration-sync` queue: +Kernel worker publishes to `integration-sync` queue: { type: "integration.connected", integration: "dropbox", account_id: "acc_xyz" } ↓ Queue consumer (apps/kernel/src/queue/integration-sync-handler.ts) receives the message ↓ -Consumer RPCs: env.KERNEL.get(idFromName("default")).runSyncAgent("dropbox", "acc_xyz") - ↓ -Kernel DO: idempotent guard — does skills/integrations/dropbox.md exist in R2? - If yes: no-op (sync agent already ran for this slug; use `/sync rerun` to force). - If no: proceed. - ↓ -Kernel DO: ensure "system" thread exists, create sync agent run row (kind="sync"), call runSubAgent({systemPrompt: SYNC_AGENT_SYSTEM_PROMPT, tools: buildSyncAgentTools(perTurn, {integration, accountId}), ...}) - ↓ -Sync agent: - Phase 1 — Triage. Is this integration sync-able (has structural surface worth mirroring)? - ├── YES → Phases 2–7 (schema, skill, handler, install, probe, verify). Writes skill md + mirror file(s) under `integrations//`, creates sync_engine row + facet. - └── NO → Phase 2-API. Writes a short API-only skill md (no sync_engine row, no facet, no mirror). Exits in ~30 seconds. - ↓ -On completion: queue ack. On failure: log + DLQ retry policy applies. +Consumer: + 1. RPC: env.KERNEL.get(idFromName("default")).skillExists("dropbox") + → if true: idempotent no-op (use `/sync rerun` to force re-run); ack message. + 2. const result = await runSyncAgent({ env, integration, accountId, kernelStub }) + 3. Ack on success; throw on failure → queue retries up to max_retries, then DLQ. ``` +`runSyncAgent` is a free function (not a class, not an RPC method) defined in `apps/kernel/src/sync/sync-agent.ts`: + +```ts +export async function runSyncAgent(args: { + env: Env; + integration: string; + accountId: string; + kernelStub: KernelStub; // for tool execution; tools RPC into the DO +}): Promise<{ ok: true; flavor: "sync-engine" | "api-only" } | { ok: false; error: string }> { + const tools = buildSyncAgentTools(args); + + const result = await streamText({ + model: openai("opus"), + system: SYNC_AGENT_SYSTEM_PROMPT, + tools, + stopWhen: stepCountIs(40), + prompt: `You are setting up sync for integration="${args.integration}", account_id="${args.accountId}". Begin Phase 1.`, + }); + + for await (const _ of result.fullStream) { + // drain the stream so the loop actually runs; nothing else listens + } + + // The agent's job is to produce side effects (skill md, sync_engine row, + // mirror files). Whether it "succeeded" = whether the skill md exists. + const ok = await args.kernelStub.skillExists(args.integration); + return ok + ? { ok: true, flavor: await args.kernelStub.skillFlavor(args.integration) } + : { ok: false, error: result.finishReason ?? "agent exited without writing skill" }; +} +``` + +That is the entire scaffolding. No thread/run/message rows. No `wrappedTool` audit either — tools mutate state via `kernelStub` RPCs that already log themselves when they touch SQLite/R2. + +**Why no thread / no run / no `runSubAgent` reuse:** the user doesn't see this agent, so there's no transcript to render, no message-by-message UI to drive, no abort cascade to wire into the TUI. `spawn_sub_agent`'s machinery (thread/run/message persistence, abort propagation, token bookkeeping for billing) exists because a sub-agent's output is visible to the user via the parent's UI. The sync agent's output IS its side effects — the skill md, the sync_engine row, the mirror files, the registered webhook. There is nothing to persist beyond those. + **Why a queue rather than direct call:** decouples connection-event detection from agent execution. The sync agent run can take minutes (sync-able branch) or 30 seconds (API-only branch) and burns tokens; running it on the connection-detection path would block other connection updates and overflow Worker CPU/wall-time limits. Queue gives us retries, DLQ for poison messages, and back-pressure for free. **Why run on every connection rather than gating by integration type:** 3,000+ Pipedream integrations and counting. Maintaining a hand-curated allowlist of "sync-able" slugs would (a) rot fast, (b) miss new integrations, (c) can't reason about *this user's* specific use of a flexible integration like Notion (could be a wiki, could be a database, could be both). The agent is the only piece capable of looking at the actual integration and deciding. The cost of running the agent on a write-only API like MailChimp is ~30 seconds of one cheap turn — small enough to absorb. -**Why a "system" thread for the run:** sync agent runs need to persist their transcript for debuggability (run row + messages, same as `spawn_sub_agent`). They don't belong in any user chat thread — they weren't typed there. v1 routes them to a special thread with id `"system"` that the kernel creates lazily. v2 may expose this via `/system` command for browsing. For v1 the user doesn't see the transcript directly; only the resulting sync_engine row, skill md, and mirror files are user-facing. +**Why Kernel DO methods at all:** the queue consumer runs in a Worker, not a DO. The sync_engine and sync_errors tables live in the Kernel DO's SQLite (single-tenant, single instance — `idFromName("default")`). So tools that mutate sync_engine, install a facet, register a webhook, or read sync state need to RPC into the Kernel DO. The kernel-side methods exist purely as RPC backends for the tools (§3.10.4 — `kernelStub.createSyncEngine(...)`, `kernelStub.updateSync(...)`, `kernelStub.runInSync(...)`, etc.). They are not "agent runtime" methods. -**Run shape:** same `runSubAgent` helper that `spawn_sub_agent` uses (extracted in PR-F2). Differences: -- `systemPrompt`: `SYNC_AGENT_SYSTEM_PROMPT` (port of sauna's, adapted for json+md output) -- `tools`: `buildSyncAgentTools(perTurn, {integration, accountId})` — curated set per §3.10.3 -- `threadId`: `"system"` (the system thread) -- `runKind`: `"sync"` (new kind alongside `main` / `sub`) -- `model`: `"opus"` (per the memory: opus default on agent-os for subagents; sync work is high-leverage and bug-sensitive) +**Debuggability without persisted transcripts:** `console.log` from the queue consumer DOES appear in `wrangler tail` (memory `feedback_wrangler_tail_misses_do_logs` only applies to DO-internal logs). The tools themselves are the source of truth: `sync_errors` rows capture handler-side failures during the run (compile errors, webhook registration failures), and the result of each tool call surfaces in the agent's own reasoning. If a run fails, re-run with `/sync rerun ` (forces idempotency bypass) and watch `wrangler tail`. No system-thread browsing UI to build. + +**Model:** `opus` (per memory `feedback_subagent_opus_default` — opus default on agent-os; sync work is high-leverage and bug-sensitive). - No `task_id` linkage (sync agent isn't running on behalf of a thread task) ### 3.10.1 Connection-event detection @@ -698,15 +723,16 @@ Two paths converge on the same outcome (queue publish): **B. Client-emitted signal (faster path).** After the user completes the Pipedream OAuth in the browser, the CLI POSTs `/integrations/connected` to the kernel with the integration slug + account_id (the CLI knows these from the Pipedream redirect). Kernel publishes the queue event immediately. This avoids the round-trip delay of waiting for the next `/connect` poll. -Both paths land at the same queue. The queue consumer is idempotent (it's gated by the existing-sync check inside `runSyncAgent`), so duplicate events from A + B are safe. +Both paths land at the same queue. The queue consumer is idempotent (gated by `kernelStub.skillExists(integration)` at the top of the consumer), so duplicate events from A + B are safe. ### 3.10.2 Failure + retry semantics Queue consumer ack semantics: - **Successful sync agent completion** (skill md written, regardless of branch) → ack. -- **Sync agent fails before writing the skill** (early discovery failure, model API error, etc.) → ack the message (do NOT retry automatically — sync agent failure is usually deterministic; user retries via `/sync rerun `). +- **Sync agent fails before writing the skill** (early discovery failure, model API error, exceeds `stopWhen` step cap, etc.) → ack the message (do NOT retry automatically — sync agent failure is usually deterministic; user retries via `/sync rerun `). - **Infrastructure failure** (kernel RPC throws, queue handler timeout) → nack → queue retries with exponential backoff. After max retries, DLQ. -- **Idempotency:** consumer checks `skills/integrations/.md` existence in R2 before starting. If skill exists, returns no-op (the agent already ran). The skill IS the trace — present skill means present decision (engine OR api-only), past trip. +- **Idempotency:** consumer checks `kernelStub.skillExists(integration)` before starting. If skill exists, no-op. The skill IS the trace — present skill means present decision (engine OR api-only), past trip. +- **Debugging a failed run:** `wrangler tail` shows the queue consumer's `console.log` output, including each tool invocation's args + result (the consumer wraps `streamText` with a small log on each tool-call event). For handler-side failures (compile errors, registerWebhook throws) `sync_errors` is the durable record. ### 3.10.3 Sync agent system prompt (sketch) @@ -858,7 +884,7 @@ The prompt cribs the sauna prompt's behavioral guidance heavily — payload-vs-c ### 3.10.4 Sync agent tool surface -Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`: +Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`. Each tool is a plain `tool()` from "ai" — NOT `wrappedTool` (no `tool_call` audit table for sync runs; the run has no thread/run row to attach audits to). Execute functions receive `{kernelStub, env}` via closure from `buildSyncAgentTools({env, integration, accountId, kernelStub})` and call kernelStub RPC methods to mutate state. Kernel-side methods log via `console.log` if needed; `wrangler tail` is the debug surface (see §3.10.2). **Read tools:** - `pipedream_fetch({host, path, method, headers, body, app_slug})` — a thin wrapper over a `fetch` that sets `X-Pd-App: ` and goes through Pipedream proxy. Returns `{status, headers, body_text}`. Same proxy `exec_code` uses. @@ -875,7 +901,7 @@ Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`: - `update_sync({sync_id, handler_js?, schema_ddl?, migration_ddl?, schema_doc?, strategy?, poll_interval_sec?, renewal_interval_sec?})` — atomic update. Bumps `version`. If `handler_js` changes, recompiles. If `schema_ddl` changes, runs `migration_ddl` against the existing facet SQLite BEFORE swapping (sauna pattern; agent provides the migration as `ALTER TABLE` / data-move statements). On migration failure, rolls back the whole update. If strategy changes from `poll` → `webhook_*`, registers webhook; vice versa unregisters. - `run_in_sync({sync_id, code})` — one-shot execution of agent-authored ES-module JS against a sync's facet (sauna pattern: separate Worker Loader bundle per call, not the persistent handler bundle). Code must export `async function runInSync(facet, fetch)` and return whatever the agent wants surfaced. `facet` exposes `__query(sql)` (read-only) and `__sqlExec(sql, ...bindings)` (writes/DDL) on the facet's SQLite, plus `__read(path)`, `__write(path, body)`, `__list(prefix?)`, `__delete(path)` on the R2 mirror (paths relative to mirror_prefix). All of this goes through the FacetBridge RpcTarget — raw DO stubs can't cross Worker Loader isolate boundaries (see sauna's run-in-sync.ts comment). `fetch` is Pipedream-proxied with the same `X-Pd-App` requirement as the handler. **This is the universal write-side channel for one-shot work**: initial backfill (Phase 5.5), diagnostics (read-only state inspection), data repair after a buggy handler version, and probe-loop triggers (firing a test event upstream so a webhook fires). Output capped at 64KB serialized JSON; logs from `console.log`/`info`/`warn`/`error` captured separately. Approval-gated. - `delete_sync({sync_id})` — destructive. Calls `handler.unregisterWebhook` if applicable, deletes `skills/integrations/.md` from R2, **list-and-deletes every key under the sync's `mirror_prefix`** (R2 batch delete), calls `ctx.facets.delete(sync.id)` (which drops the facet's SQLite — `raw_events`, `agent_dedup`, everything goes), flips sync_engine row to status='deleted' (kept for audit). For Branch-B (api-only) syncs there's no sync_engine row to flip — `delete_sync` just removes the skill. -- `manage_tasks` — existing. The sync agent uses it to track its own phases (good UX: "Phase 1: discovery 🔄 Phase 2: schema 🔄 ..."). +- (No `manage_tasks`. The agent has no UI; it tracks phases in its own reasoning context. `manage_tasks` writes to a per-thread/per-run tasks table; the sync agent has neither.) Tools the sync agent does **not** get: - `exec_code` — the sync agent's JS authoring happens through `create_sync_engine` and `run_in_sync`; there's no reason to expose raw exec_code (smaller blast radius, clearer audit trail). @@ -913,24 +939,21 @@ CLI → Kernel API: POST /integrations/connected { integration: "dropbox", accou ↓ Kernel: publish to env.INTEGRATION_SYNC queue: { type: "integration.connected", integration: "dropbox", account_id: "acc_xyz" } ↓ -Queue Consumer: receives message → env.KERNEL.get(idFromName("default")).runSyncAgent("dropbox", "acc_xyz") - ↓ -Kernel DO: idempotency check on sync_engine for (dropbox, acc_xyz) → none exists, proceed -Kernel DO: ensure "system" thread exists; INSERT run row (kind="sync") in system thread -Kernel DO: runSubAgent({systemPrompt: SYNC_AGENT_SYSTEM_PROMPT, tools: buildSyncAgentTools(perTurn, {integration:"dropbox", accountId:"acc_xyz"}), prompt: "", model: "opus", threadId: "system", runKind: "sync"}) +Queue Consumer: receives message ↓ - [sync agent run begins, transcripts persist in system thread for debugging] +Consumer: const kernelStub = env.KERNEL.get(idFromName("default")); +Consumer: if (await kernelStub.skillExists("dropbox")) { ack; return; } +Consumer: await runSyncAgent({ env, integration:"dropbox", accountId:"acc_xyz", kernelStub }); + // streamText runs IN-PROCESS in the consumer. No thread row, no run row, no message row. + // Tool execution is the only thing that crosses into the Kernel DO (via kernelStub). ↓ -SyncAgent: manage_tasks create "Sync dropbox" SyncAgent: fetch_docs https://www.dropbox.com/developers/reference/webhooks SyncAgent: pipedream_fetch dropbox /2/users/get_current_account → {account_id, name, country, ...} SyncAgent: pipedream_fetch dropbox /2/files/list_folder { path: "" } → top-level entries -SyncAgent: (recurse breadth-first, build mental model) -SyncAgent: manage_tasks update "Sync dropbox: Schema" -SyncAgent: (decide on flat-entries shape; write schema_doc) -SyncAgent: manage_tasks update "Sync dropbox: Skill" +SyncAgent: (recurse breadth-first, build mental model — all in the agent's reasoning context) +SyncAgent: (decide on flat-entries shape; write schema_doc inline) SyncAgent: write_skill { slug: "dropbox", body: "" } -SyncAgent: manage_tasks update "Sync dropbox: Handler" + → tool execute: kernelStub.writeSkill("dropbox", body) → R2 PUT SyncAgent: (write handler_js: verifySignature, handleWebhook (calls list_folder/continue with cursor), registerWebhook, unregisterWebhook — no backfill export) SyncAgent: create_sync_engine { integration: "dropbox", @@ -941,18 +964,19 @@ SyncAgent: create_sync_engine { skill_path: "skills/integrations/dropbox.md" // mirror_prefix omitted → defaults to "integrations/dropbox/" } -Kernel: INSERT sync_engine row, status='discovering' -Kernel: getOrCreateFacet → ctx.facets.get(sync.id, () => { ...buildHandlerBundle → getDurableObjectClass("Handler") }). Wrapper constructor runs. + → tool execute: kernelStub.createSyncEngine(...) +Kernel DO: INSERT sync_engine row, status='discovering' +Kernel DO: getOrCreateFacet → ctx.facets.get(sync.id, () => { ...buildHandlerBundle → getDurableObjectClass("Handler") }). Wrapper constructor runs. Wrapper: ctx.storage.sql.exec(sync.schema_ddl) // runs CREATE TABLE IF NOT EXISTS ... for facet's own SQLite - (On Worker Loader compile failure OR DDL syntax error: rollback row, errorText, ctx.facets.delete to clean up partial facet, return error.) -Kernel: recomputeNextAlarm() // supervisor recalculates the min next-alarm across all active syncs and calls this.ctx.storage.setAlarm(at) on itself -Kernel: await facet.registerWebhook({callbackUrl: "https:///webhook/dropbox/acc_xyz", label: "agent-os dropbox sync"}) + (On Worker Loader compile failure OR DDL syntax error: rollback row, errorText, ctx.facets.delete to clean up partial facet, return error to the tool — sync agent sees the failure in its next reasoning step.) +Kernel DO: recomputeNextAlarm() // supervisor recalculates the min next-alarm across all active syncs and calls this.ctx.storage.setAlarm(at) on itself +Kernel DO: await facet.registerWebhook({callbackUrl: "https:///webhook/dropbox/acc_xyz", label: "agent-os dropbox sync"}) → returns {id: "dbid_123", secret: "...", expiresAt: null} -Kernel: UPDATE sync_engine SET webhook_id, webhook_secret, webhook_expires_at -Kernel: UPDATE sync_engine SET status='active' - (Supervisor recomputeNextAlarm() — dropbox has no expiry → no alarm contribution.) +Kernel DO: UPDATE sync_engine SET webhook_id, webhook_secret, webhook_expires_at +Kernel DO: UPDATE sync_engine SET status='active' + (recomputeNextAlarm() — dropbox has no expiry → no alarm contribution.) + → tool returns { ok: true, sync_id } to the sync agent ↓ -SyncAgent: manage_tasks update "Sync dropbox: Backfill" SyncAgent: run_in_sync { sync_id, code: "export async function runInSync(facet, fetch) { @@ -979,17 +1003,19 @@ SyncAgent: run_in_sync { return { entries: entries.length }; }" } -Kernel: load exec bundle via env.LOADER.load(buildExecWorkerCode({ code, ... })) — separate from the handler bundle + → tool execute: kernelStub.runInSync(sync_id, code) +Kernel DO: load exec bundle via env.LOADER.load(buildExecWorkerCode({ code, ... })) — separate from the handler bundle → calls FacetBridge.runInSync(facet, fetch) → exec returns { ok: true, result: { entries: 1247 }, logs: [...] } + → tool returns the result to the agent SyncAgent: (sees {entries: 1247}; tree.json is now populated) -SyncAgent: manage_tasks update "Sync dropbox: Verify" SyncAgent: list_mirror sync_id → ["tree.json"] SyncAgent: read_mirror sync_id, "tree.json" → confirms shape SyncAgent: get_sync_errors sync_id → empty -SyncAgent: manage_tasks complete "Sync dropbox: Done" +SyncAgent: streamText finishes (model emits stop signal — no further tools called) ↓ - [final answer persists as the last message of the sync run in the system thread; queue consumer acks the message] +Consumer: runSyncAgent returns { ok: true, flavor: "sync-engine" }; queue message ack. + (Nothing else persists. The skill md + sync_engine row + facet + tree.json + registered webhook ARE the deliverable.) ↓ [next time the user types in any thread, dropbox is in , the agent reads skills/integrations/dropbox.md when relevant] ``` @@ -1135,19 +1161,18 @@ exec_code returns success. ### PR-F2 — Queue trigger + sync agent + Dropbox e2e **Files created:** - `apps/kernel/src/sync/sync-agent-system-prompt.ts` — ported from sauna -- `apps/kernel/src/sync/sync-agent-tools.ts` — `buildSyncAgentTools(perTurn, {integration, accountId})` returning the 9 curated tools -- `apps/kernel/src/queue/integration-sync-handler.ts` — Cloudflare Queue consumer; receives `integration.connected` messages, RPCs `kernel.runSyncAgent` +- `apps/kernel/src/sync/sync-agent-tools.ts` — `buildSyncAgentTools({env, integration, accountId, kernelStub})` returning the curated tools (plain `tool()` from "ai", no `wrappedTool`) +- `apps/kernel/src/sync/sync-agent.ts` — `runSyncAgent({env, integration, accountId, kernelStub})` free function; wraps `streamText` with the prompt + tools + opus model +- `apps/kernel/src/queue/integration-sync-handler.ts` — Cloudflare Queue consumer; receives `integration.connected` messages, builds `kernelStub`, calls `runSyncAgent` directly (no DO RPC needed — runSyncAgent runs IN the consumer) - `apps/kernel/src/api/integrations-connected.ts` — Hono route `POST /integrations/connected` (CLI-emitted signal path); validates body, publishes to queue -- `apps/kernel/src/sync/run-sub-agent.ts` — extracted helper (factored out of `apps/kernel/src/tools/spawn-sub-agent.ts`'s execute body) **Files modified:** -- `apps/kernel/src/tools/spawn-sub-agent.ts` — refactor to use the extracted `runSubAgent` -- `apps/kernel/src/kernel.ts` — add `runSyncAgent(integration, accountId)` RPC method; ensure "system" thread row exists lazily +- `apps/kernel/src/kernel.ts` — add the RPC methods the sync-agent tools need as backends: `skillExists(slug)`, `skillFlavor(slug)`, `writeSkill(slug, body)`, `createSyncEngine(...)`, `updateSync(...)`, `runInSync(syncId, code)`, `deleteSync(syncId)`, `readMirror(syncId, path)`, `listMirror(syncId, prefix?)`, `querySync(syncId, sql)`, `getSyncErrors(syncId, limit?)`. **NO `runSyncAgent` RPC method** — the sync agent runs in the consumer, not the DO. **NO system-thread machinery.** **NO `runSubAgent` extraction** — `spawn_sub_agent` is unrelated and stays as-is. - `apps/kernel/src/integrations/sync.ts` — extend `diffAccounts()` callsite to publish `integration.connected` events to the queue (kernel-poll path; complement to the CLI-emitted path) -- `apps/cli/src/commands/sync.ts` — add `/sync rerun ` — POSTs to `/integrations/connected` with `force: true` bit that bypasses the idempotency check inside `runSyncAgent` +- `apps/cli/src/commands/sync.ts` — add `/sync rerun ` — POSTs to `/integrations/connected` with `force: true` bit so the consumer skips the `kernelStub.skillExists` short-circuit - `wrangler.jsonc` — declare `integration-sync` queue (producer + consumer bindings; DLQ; max retries) -**Smoke:** complete `/connect dropbox` in CLI (or simulate by direct POST to `/integrations/connected`) → queue receives event → consumer fires → sync agent run begins in system thread → triage decides Branch A → ~5 min later, `skills/integrations/dropbox.md` (sync-engine flavor) and the agent-chosen file(s) under `integrations/dropbox/` exist; main agent (in any thread) sees `dropbox` in `` and reads the skill; ask "where are my tax filings?" → answer correct; main agent uses `exec_code` to upload a file → webhook fires → mirror reflects within 30s; `/sync delete dropbox` cleans up upstream + every R2 key under the prefix. Then ALSO smoke Branch B: connect MailChimp → queue → triage decides Branch B → ~30s later, `skills/integrations/mailchimp.md` (api-only flavor) exists, no sync_engine row, no files under `integrations/mailchimp/`. +**Smoke:** complete `/connect dropbox` in CLI (or simulate by direct POST to `/integrations/connected`) → queue receives event → consumer fires → `runSyncAgent` runs in-process → triage decides Branch A → ~5 min later, `skills/integrations/dropbox.md` (sync-engine flavor) and the agent-chosen file(s) under `integrations/dropbox/` exist; main agent (in any thread) sees `dropbox` in `` and reads the skill; ask "where are my tax filings?" → answer correct; main agent uses `exec_code` to upload a file → webhook fires → mirror reflects within 30s; `/sync delete dropbox` cleans up upstream + every R2 key under the prefix. Then ALSO smoke Branch B: connect MailChimp → queue → triage decides Branch B → ~30s later, `skills/integrations/mailchimp.md` (api-only flavor) exists, no sync_engine row, no files under `integrations/mailchimp/`. **Debug surface during the smoke:** `wrangler tail` shows each tool call's args + result from the consumer's `console.log`. ### PR-F3 — Polling fallback + research wave + additional integrations **Files created:** @@ -1202,7 +1227,7 @@ exec_code returns success. ### PR-F2 smoke 1. From zero state (no fixture, no sync_engine row), simulate the connection event: `curl -X POST http://localhost:8787/integrations/connected -d '{"integration":"dropbox","account_id":"acc_xyz"}'`. -2. Queue receives the message; consumer fires; sync agent run begins (visible as a new run row in the `system` thread, queryable via DB). +2. Queue receives the message; consumer fires; `runSyncAgent` begins (visible in `wrangler tail` as the consumer's `console.log` output streams tool calls). 3. Sync agent completes within ~5–10 min (target; cost ceiling enforced). 4. `skills/integrations/dropbox.md` exists in R2 and accurately describes the test account's folder structure in prose. 5. `integrations/dropbox/tree.json` (or whatever filename the agent chose — likely tree.json given §3.4 example) exists, has the expected shape, includes top-level folders. `list_mirror` over the prefix returns the files the skill's "Mirror layout" section claims will be there. @@ -1239,8 +1264,8 @@ exec_code returns success. | 7b | Sync agent runs on EVERY connection; triages first | 3000+ Pipedream integrations; most are API-wrappers with nothing to sync. Hand-curated allowlist would rot fast. Agent's first phase decides "Branch A (full sync engine)" vs "Branch B (skill-only)". Branch B costs ~30s of one cheap run; small enough to absorb. | | 7c | NO `` injected block; rely on existing `` + skill convention | Pre-injecting metadata for N connected integrations is context bloat at scale. The skill md file IS the description — the agent reads it lazily when the question touches that integration. One system-prompt sentence teaches the convention. | | 7d | Skill md is the universal contract (sync-engine flavor OR api-only flavor) | Every connected integration gets a `skills/integrations/.md`. The skill's frontmatter `type` says which flavor. Idempotency check uses skill-existence. One unified surface for the main agent regardless of how rich the underlying sync is. | -| 8 | Sync agent transcripts persist in a dedicated `"system"` thread | Sync runs need debuggable transcripts (same as `spawn_sub_agent`) but don't belong in any user chat. v1: kernel creates a `"system"` thread lazily and writes there. v2: surface via `/system` browse command. | -| 9 | `runSubAgent` helper extracted from `spawn_sub_agent`'s body; reused by both `spawn_sub_agent` tool AND the queue consumer | Different system prompt + curated tool surface for each caller; only one piece of shared machinery (run/message persistence, abort cascade, token bookkeeping). | +| 8 | Sync agent has **no thread, no run row, no message persistence, no streaming, no audit table** | Per user direction. The agent runs as a plain AI-SDK `streamText` invocation inside the queue consumer. The user never sees it; there's nothing to render. Its output IS its side effects (skill md + sync_engine row + facet + mirror files + registered webhook). `spawn_sub_agent`'s thread/run/message machinery exists only because sub-agents are visible to the user via the parent UI; the sync agent is not. Debuggability comes from `wrangler tail` on the consumer's `console.log` output plus the durable `sync_errors` table. | +| 9 | No `runSubAgent` extraction; `spawn_sub_agent` and `runSyncAgent` share nothing | Different runtimes (sub-agent runs inside the Kernel DO with full thread/run/message machinery; sync agent runs in a queue-consumer Worker with none of it). Forcing a shared helper would either over-abstract or pull thread/run plumbing into the consumer where it has no purpose. Each tool stays focused on its caller. | | 10 | Dropbox is the v1 e2e smoke target despite being archetype B | Matches the user's VC-example demo. Channel + cursor + empty-ping primitives are needed in F1/F2 anyway since Dropbox uses them. | | 11 | 3-PR stack (not 4) | Foundation primitives + queue trigger + agent + dropbox e2e split as F1/F2; F3 covers the other archetypes (Linear stable, gcal channel, Notion poll-or-webhook-bus). | | 12 | Master spec + per-PR plans | One coherent architecture doc; per-PR plans land when each PR kicks off so they reflect the implementation reality. | From fe8a18ea709742f4a0aadad3c594997b4010c64d Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 11:24:18 +0530 Subject: [PATCH 06/13] docs(sync-agent): main-agent R2 read primitives + NDJSON-friendly mirror layout MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Mirror-as-index only works if the main agent can read mirror files without dragging the whole object into its context window. Adds two main-agent tools and threads matching guidance through the sync agent's design rules. §3.1.1 (new): Main-agent R2 read primitives - read_file extended with offset/limit (1-based lines) + 64KB soft cap on full reads. Cap-violation error points at grep_file / paginated reads. - grep_file (new): server-side regex over R2 files with path or prefix+glob scope. Plain JS RegExp (no WASM ripgrep). Returns {file, line, match, before[], after[]} records, capped at 100 matches. - Both wrappedTool({touchesFS:false}) per the read-only-tools convention. - Argued out: two tools beat one composite (divergent return shapes, arg surfaces, failure hints; Claude Code precedent). §3.4: mirror layout gets a budget rule - Aim for ≤64KB per file so read_file is one-shot. - For high-cardinality indexes: NDJSON (line-grep-friendly, line-offset-readable) or sharding (per-axis files + a top-level index.json). Mix is fine (Gmail = labels.json + per-label NDJSON). - Document the layout in the skill md so the main agent picks read_file vs grep_file correctly. §3.4 Dropbox canonical example switched to NDJSON+meta: - tree.ndjson (one entry per line) + meta.json (cursor + counts). - §3.5.1 skill md example, §3.10.3 Phase 2 schema design, §4.1 trace (run_in_sync code), §4.2 webhook delivery handler, §4.3 main agent query (grep_file instead of read_file), §6 PR-F1 fixture line, §6 PR-F2 smoke checklist — all updated to match. §6 PR slicing: - New PR-R1 (Main-agent R2 read primitives) — independent prerequisite, no compile-time dependency on F1/F2/F3 but unblocks their design assumptions. Files modified: read-file.ts. Files created: grep-file.ts. 4 smoke checks. §10 decisions: - Row 6: updated wording (meta.json/tree.ndjson pattern; "never embed a churning cursor in a multi-MB index file"). - Row 14 (new): read_file + grep_file, not composite — divergent shapes + Claude Code precedent. - Row 15 (new): 64KB cap + NDJSON/sharding preference; guardrail with hint, not hard refusal (offset/limit bypasses). Co-Authored-By: Claude Opus 4.7 (1M context) --- .../specs/2026-05-17-sync-agent-design.md | 146 +++++++++++++----- 1 file changed, 110 insertions(+), 36 deletions(-) diff --git a/docs/superpowers/specs/2026-05-17-sync-agent-design.md b/docs/superpowers/specs/2026-05-17-sync-agent-design.md index c7ef1a6..86acaf9 100644 --- a/docs/superpowers/specs/2026-05-17-sync-agent-design.md +++ b/docs/superpowers/specs/2026-05-17-sync-agent-design.md @@ -101,6 +101,30 @@ The main agent's runtime flow for a question that touches a connected integratio Why no injected block: 3,000+ Pipedream integrations means most users will end up with many connected slugs. Pre-injecting metadata for all of them is context bloat. The skill file *is* the authoritative description and the agent loads it lazily, only when it's relevant to the current turn. +### 3.1.1 Main-agent read primitives over R2 + +The architecture only works if the main agent can read mirror files without dragging the whole object into its context window. A 3MB `tree.json` blown into context to answer "where's my 2024 tax form?" defeats the entire mirror-as-index premise. The main agent therefore gets two focused R2 read tools — `read_file` (extended) and `grep_file` (new) — and the sync agent gets matching prompt guidance to write files those tools can do well against. + +These are **generic R2 FS tools**, not mirror-specific. They serve skill files, the mirror, anything we ever put in R2. They are independent of the sync agent work — usable from any agent surface that reads R2. + +**`read_file({ path, offset?, limit? })`** — extends the existing tool with line-based pagination and a soft cap: + +- `path` is an R2 key (no prefix scoping; main agent already knows the full path from the skill md). +- `offset` (1-based line number) + `limit` (line count, default 2000) work the same way Claude Code's Read does — returns text with a line-number gutter so the agent can address sub-ranges in follow-up calls. +- **Soft cap: 64KB.** Above the cap (and no `offset`/`limit` provided), the tool errors with: `file is N bytes; use grep_file for substring search or call read_file with {offset, limit} for chunked reads`. The cap is a guardrail against the agent accidentally torching its own context; with explicit `offset`/`limit` the cap is bypassed (the agent has asked for a specific slice on purpose). 64KB is the right number because R2 reads in a Worker are full-buffer — there's no streaming partial-content shortcut — so the cap doubles as a Worker CPU/memory guardrail. + +**`grep_file({ pattern, path?, prefix?, glob?, regex_flags?, context_lines?, max_matches? })`** — new tool, server-side regex over R2 files: + +- Scope: exactly one of `path` (single file) OR `prefix` + optional `glob` (`*.json`, `tree/by-*.ndjson`, etc.) — recurses under the prefix, filters by glob, runs the pattern across each matched file. +- `pattern` is a regex string compiled with plain JS `RegExp` (no WASM ripgrep — JS regex is fast enough for our file sizes and one fewer dep). `regex_flags` defaults to `"m"` (multiline) so `^` / `$` are line-anchored; agent can pass `"mi"` for case-insensitive, etc. +- `context_lines` defaults to 2 (lines of context before/after each match, ripgrep-style). +- `max_matches` defaults to 100 (across all files in scope). Past the cap the tool returns the first 100 with `truncated: true` so the agent knows to narrow. +- Returns `Array<{ file, line, match, before: string[], after: string[] }>`. The 3MB R2 object stays in the Worker; only matching slices cross into agent context. This is the **primary** read path for any non-trivial mirror — `read_file` is the fallback when the agent already knows the exact slice. + +**Why two tools, not one composite:** return shapes diverge (text vs match records), argument surfaces diverge (offset/limit vs pattern/glob/context), and the failure-recovery hints diverge (read says "use grep_file", grep says "narrow the regex / scope prefix"). Folding both into one schema produces a union return type the agent has to discriminate on every call, and LLMs are measurably worse on union returns than focused ones. Claude Code keeps Read and Grep separate for the same reasons; we copy the precedent. + +**Where it lives:** `apps/kernel/src/tools/read-file.ts` (extend) and `apps/kernel/src/tools/grep-file.ts` (new). These are wrapped via `wrappedTool({ touchesFS: false })` per memory `feedback_wrappedtool_default_readonly` — read-only main-agent tools belong in the `tool_call` audit table. Both ship in their own PR (§6 PR-R1) — independent of the sync-agent stack but prerequisite for it to function as designed. + ### 3.2 The three webhook archetypes The sync agent must support all three: @@ -129,7 +153,7 @@ export const syncEngine = sqliteTable("sync_engine", { schemaDdl: text("schema_ddl").notNull().default(""), // SQL DDL for the facet's own SQLite (dedup tables, raw_events for probe loop, cursor state, etc.). Empty string allowed for handlers that don't need any SQL state. Run by the supervisor at facet bootstrap. schemaDoc: text("schema_doc").notNull(), // free-form prose describing the mirror layout: which files live under the prefix, what shape each one has, what's in them vs. fetched on demand skillPath: text("skill_path").notNull(), // "skills/integrations/dropbox.md" - mirrorPrefix: text("mirror_prefix").notNull(), // R2 key prefix, e.g. "integrations/dropbox/" — defaults to `integrations//`. Every mirror read/write/list/delete from the facet is scoped under this prefix. The agent decides whether the layout is one file (tree.json) or many (labels.json, threads/by-label/inbox.json, ...); the supervisor never inspects the contents. + mirrorPrefix: text("mirror_prefix").notNull(), // R2 key prefix, e.g. "integrations/dropbox/" — defaults to `integrations//`. Every mirror read/write/list/delete from the facet is scoped under this prefix. The agent decides the layout — one JSON file, many JSON files, NDJSON, sharded prefixes; the supervisor never inspects the contents. // Webhook state webhookId: text("webhook_id"), // upstream's id for the registered webhook (for unregister) webhookSecret: text("webhook_secret"), // for verifySignature @@ -179,7 +203,7 @@ DESC index on `(sync_id, occurred_at)` supports `SELECT ... ORDER BY occurred_at **Cursors and webhook state live wherever the agent decides.** The supervisor doesn't care. Three first-class options the system prompt will teach (§3.10.3); the agent picks per integration: -- **Embedded in a mirror file** (`tree.json#_cursor`). Best when the cursor is low-churn AND atomic with the file it gates — Dropbox's single tree, Airtable's schema-of-schemas. +- **Embedded in a mirror file** (`meta.json#_cursor` next to a `tree.ndjson`, or `workspace.json#_cursor` for bounded shapes). Best when the cursor is low-churn AND atomic with the data it gates. The cursor file should always be the small one — never the megabyte-sized index — so cursor writes don't trigger huge rewrites. - **Sidecar mirror file** (`cursors.json`). Best when cursor churn is medium AND the main index is large enough that you don't want to rewrite it on every cursor bump. - **Facet SQLite row** (`webhook_config` table: `channel_id`, `expires_at`, `history_id`, `last_seen_at`, ...). Best when state churns per-webhook OR there's more than just a cursor (registration ids, expiry, per-resource pointers). A single `UPDATE webhook_config SET history_id=?, last_seen_at=?` on every webhook beats `read JSON → parse → mutate → stringify → R2 write` — by a lot. Gmail's per-label history pointers and Drive's per-folder channel state are the canonical fits. @@ -205,21 +229,29 @@ All `path` arguments are **relative to `mirror_prefix`** — the supervisor's `M **Why prefix-scoped instead of a single file:** - Concurrency: webhook handlers can update one file (e.g. a per-label index) without rewriting the whole tree. Single-file mirrors force whole-tree rewrites on every delta and serialize writes through R2's strong-consistency layer. -- Size: a Gmail mirror that flattens everything into one json bloats fast. Splitting by label / by-page keeps each file small enough to grep with `read_file` without overflowing the main agent's context window. -- Locality: main agent finds what it needs faster when the layout itself encodes structure. Reading `integrations/gmail/threads/by-label/inbox.json` is faster than greping a giant `gmail.json` for `label: "INBOX"`. +- Size: a Gmail mirror that flattens everything into one json bloats fast. Splitting by label / by-page keeps each file inside the main agent's `read_file` 64KB soft cap (§3.1.1) so a query doesn't have to fall back to `grep_file`. +- Locality: main agent finds what it needs faster when the layout itself encodes structure. Reading `integrations/gmail/threads/by-label/inbox.json` is faster than `grep_file`-ing a giant `gmail.json` for `label: "INBOX"`. - Cost: R2 charges per operation. Partial updates win when only a slice of the index changes. +**Design with the main-agent read primitives in mind (§3.1.1).** Two rules of thumb for the sync agent, taught in the prompt (§3.10.3 Phase 2), not enforced by the runtime: + +- **Aim for ≤64KB per file.** That's `read_file`'s soft cap. Files under it land in a single full read; over it, the main agent must `grep_file` or paginate. Neither is wrong, but a sub-cap full read is the fastest path. Bounded-shape indexes (Airtable schemas, Notion top-level page list, Slack channel list) fit comfortably; high-cardinality indexes don't. +- **For high-cardinality indexes, prefer NDJSON or shard the layout.** Newline-delimited JSON (one entry per line) is line-grep-friendly, line-offset-readable, and append-friendly. `grep_file` is line-oriented so NDJSON costs nothing on the read side — and the layout naturally caps memory per match. Sharding is the alternative: split by a natural axis (per-label, per-base, per-repo) plus a top-level `index.json` advertising the shards so the main agent reads the small index, finds the right shard, then `grep_file` or `read_file`s that one. Both are valid; NDJSON tends to win for flat lists (Dropbox files, Gmail thread headers, Linear issue headers); sharding tends to win when the axis itself is what the main agent queries against (Gmail by-label, GitHub by-repo). + +The skill md's "Mirror layout" section (§3.5) MUST advertise whether files are JSON, NDJSON, or sharded — the main agent uses that to pick `read_file` vs `grep_file` vs "read the index first" on each query. + **The agent designs the layout and documents it in the skill md.** That layout *is* the schema_doc. Examples: -**Dropbox (file tree) — single file is fine.** Whole tree is small, all-or-nothing replacement on each empty-ping delta is cheap. +**Dropbox (file tree) — single NDJSON file, cursor in a separate small JSON.** Most user Dropboxes have hundreds to tens of thousands of files; one fat `tree.json` quickly exceeds the 64KB read cap. NDJSON entries + a tiny `meta.json` for the cursor keeps the index `grep_file`-friendly without forcing the agent to pick shards. ``` integrations/dropbox/ -└── tree.json # { _cursor, _updated_at, entries: [{id, path, name, type, size, modified, mime}, ...] } +├── meta.json # { _cursor, _updated_at, entry_count } ← always small; full read every time +└── tree.ndjson # one line per entry: {"id":"id:abc","path":"/Personal/Tax/2024-form-1040.pdf","name":"...","type":"file","size":142331,"modified":"...","mime":"application/pdf"} ``` `schema_doc`: -> Single `tree.json` at the prefix root. `entries` is a flat array of `{id, path, name, type, size, modified, mime}`. `_cursor` is Dropbox's list_folder/continue cursor. File contents NOT mirrored; download via `https://content.dropboxapi.com/2/files/download` with `X-Pd-App: dropbox` from `exec_code`. +> `meta.json` holds the cursor and counts. `tree.ndjson` holds one entry per line: `{id, path, name, type, size, modified, mime}`. Main agent finds tax filings with `grep_file({pattern: "^.*\\"path\\":\\"/Personal/Tax/", path: "integrations/dropbox/tree.ndjson"})` — only the matching lines come back, regardless of how many files are in the tree. For a small subset by id, `read_file` with a known line range works too. File contents NOT mirrored; download via `https://content.dropboxapi.com/2/files/download` with `X-Pd-App: dropbox` from `exec_code`. **Gmail (high-volume, multiple useful slices) — multi-file mirror + SQLite cursor pays off.** @@ -295,7 +327,8 @@ generated_at: 2026-05-17T18:30:12Z ## Mirror layout -Single file: `integrations/dropbox/tree.json`. Shape: `{ _cursor, _updated_at, entries: [{id, path, name, type, size, modified, mime}, ...] }`. See schema_doc below for full detail. +- `integrations/dropbox/meta.json` — `{ _cursor, _updated_at, entry_count }`. Small (always full-readable). +- `integrations/dropbox/tree.ndjson` — one entry per line: `{id, path, name, type, size, modified, mime}`. Line-oriented; use `grep_file` for any path-based query. [schema_doc embedded here] @@ -318,15 +351,18 @@ Everything lives under two roots: ## What's in the mirror vs. fetched on demand -In the mirror (`integrations/dropbox/tree.json`): file/folder paths, names, ids, types, sizes, modified-at, mime-types. **NOT** file contents. +In the mirror (`tree.ndjson` + `meta.json`): file/folder paths, names, ids, types, sizes, modified-at, mime-types. **NOT** file contents. To get a file's contents, `exec_code` the Dropbox download API (snippet under "Reading content" below). ## Query patterns (mirror-only — no API call needed) -- Find tax filings: read `integrations/dropbox/tree.json`, filter `entries` where `path` starts with `/Personal/Tax/` -- Find a portfolio company's docs: filter `path` matches `/Fund//Portfolio//` -- Find the most recent board deck for a company: filter `path` matches `/Fund/*/Portfolio//Board/`, sort by `modified` desc +Use `grep_file` against `tree.ndjson`; each match is one entry-line of JSON: + +- Find tax filings: `grep_file({ pattern: "\\"path\\":\\"/Personal/Tax/", path: "integrations/dropbox/tree.ndjson" })` +- Find a portfolio company's docs: `grep_file({ pattern: "\\"path\\":\\"/Fund/[^/]+/Portfolio/AcmeCorp/", path: "integrations/dropbox/tree.ndjson" })` +- Find the most recent board deck for a company: same `grep_file` (`Board/` segment), then sort the returned matches by `modified` desc in the agent's reasoning. +- Inspect the cursor / freshness: `read_file integrations/dropbox/meta.json`. ## Reading content (mirror gives the path; exec_code fetches the bytes) @@ -527,7 +563,7 @@ interface HandlerCtx { That's it. No supervisor SQL access (only facet's own sqlite). No sub-agent spawning. No process or filesystem access beyond mirror RPCs. The handler is a pure async function with fetch + mirror primitives + scratch sqlite. -**Why this is enough for Dropbox:** the handler `mirror.read("tree.json")` (parses it to get `_cursor`), calls `https://api.dropboxapi.com/2/files/list_folder/continue` with the cursor, receives a list of file changes, applies them to `entries`, updates `_cursor`, `mirror.write("tree.json", JSON.stringify(next))`. ~30 lines of JS. +**Why this is enough for Dropbox:** the handler `mirror.read("meta.json")` (small file, parses to get `_cursor`), calls `https://api.dropboxapi.com/2/files/list_folder/continue` with the cursor, receives a list of file changes. For each change: if added/modified, find the line in `tree.ndjson` and replace (or append); if deleted, remove the line. Write `tree.ndjson` back, write `meta.json` back with the new cursor. ~40 lines of JS. (NDJSON write costs the same as JSON write — full rewrite per webhook — but the read side benefits the main agent's grep, and that's the hot path.) **Why this is enough for Gmail:** handler receives a webhook for label X, does `SELECT history_id FROM webhook_config` via `ctx.sql` (single SQL read), fetches the history slice via the Gmail API, `mirror.read("threads/by-label/.json")`, applies the delta to that slice only, writes it back, runs `UPDATE webhook_config SET history_id = ?, last_seen_at = ?` (single SQL write). Two R2 ops per webhook (one read + one write of the one changed label file) and two SQL ops — vs three R2 ops if the cursor lived in `cursors.json`. The unchanged label files stay untouched either way. @@ -796,7 +832,15 @@ THE PRIME DIRECTIVE — re-read before designing the schema: > > Sanity check: imagine the user has 50,000 records / 100GB of files / 10,000 issues. Does your mirror stay under ~5MB? If not, you're mirroring content. -Decide the mirror layout based on what discovery surfaced, applying the Prime Directive ruthlessly: how many files, named what, with what shape (see §3.4 examples — Dropbox = single tree.json; Gmail = labels/cursors/per-label-threads). Write the `schema_doc` prose. The schema_doc must list every file the mirror will contain, the shape of each, AND explicitly call out "what's in the mirror vs. fetched on demand" so the main agent knows both halves. +Decide the mirror layout based on what discovery surfaced, applying the Prime Directive ruthlessly: how many files, named what, with what shape (see §3.4 examples — Dropbox = meta.json + tree.ndjson; Gmail = labels.json + per-label thread files + SQLite cursor state). Write the `schema_doc` prose. The schema_doc must list every file the mirror will contain, the shape of each, AND explicitly call out "what's in the mirror vs. fetched on demand" so the main agent knows both halves. + +**Budget rule** (taught here, not enforced by the runtime): aim for each individual mirror file ≤64KB so the main agent's `read_file` returns it in one shot (§3.1.1). For high-cardinality indexes that exceed that, the two patterns are: +- **NDJSON** (one entry per line) when the index is a flat list and the main agent's typical query is a substring/regex over an entry field. `grep_file` is line-oriented and returns only matching entries — context budget tracks matches, not file size. Dropbox file trees, Gmail thread headers, Linear issue headers all fit this. +- **Sharding** (split by a natural axis + a top-level `index.json` listing the shards) when the axis itself is what the main agent queries against. Gmail by-label, GitHub by-repo, Notion by-database. The main agent reads the small index file, finds the right shard, then `read_file` or `grep_file` that one. + +You can mix: Gmail's `labels.json` (bounded JSON) + `threads/by-label/.ndjson` (per-label NDJSON shards) covers both axes. + +Document the choice in the skill md's "Mirror layout" section — flag JSON vs NDJSON vs sharded so the main agent picks `read_file` vs `grep_file` correctly on each query. Then decide the **facet SQLite schema** — separate from the R2 mirror. Write `schema_ddl` as a SQL string with `CREATE TABLE IF NOT EXISTS` statements. This is the handler-internal state layer; not visible to the main agent. Typical contents: @@ -829,7 +873,7 @@ CREATE TABLE IF NOT EXISTS cursors ( ); -- For low-churn cursors atomic with a small mirror, skip SQL entirely and --- embed the cursor in the mirror file (e.g. tree.json#_cursor for Dropbox). +-- embed the cursor in a small mirror file (e.g. meta.json#_cursor next to a tree.ndjson). -- For the probe-loop's phase-1 raw capture (Phase 6): CREATE TABLE IF NOT EXISTS raw_events ( @@ -890,7 +934,7 @@ Curated, defined in `apps/kernel/src/sync/sync-agent-tools.ts`. Each tool is a p - `pipedream_fetch({host, path, method, headers, body, app_slug})` — a thin wrapper over a `fetch` that sets `X-Pd-App: ` and goes through Pipedream proxy. Returns `{status, headers, body_text}`. Same proxy `exec_code` uses. - `web_search(query)` — existing. - `fetch_docs(url)` — existing (Firecrawl-backed). -- `read_mirror(sync_id, path)` — reads one file from a sync's mirror prefix. `path` is relative (e.g. `tree.json`, `threads/by-label/INBOX.json`). Returns the body as a string or `null` if missing. +- `read_mirror(sync_id, path)` — reads one file from a sync's mirror prefix. `path` is relative (e.g. `meta.json`, `tree.ndjson`, `threads/by-label/INBOX.ndjson`). Returns the body as a string or `null` if missing. Sync agent has no size cap here — verify steps in Phase 7 sometimes need the full file. (Main agent's `read_file` is the one with the 64KB cap; that's §3.1.1.) - `list_mirror(sync_id, prefix?)` — lists keys under the sync's mirror prefix (further filtered by an optional sub-prefix). Used during verify (`Phase 7`) to confirm the layout the agent wrote actually exists, and during debugging. - `query_sync(sync_id, sql)` — SELECT-only SQL against the facet's own SQLite. Used to read `raw_events` during the probe loop, inspect `agent_dedup` state during debugging, etc. Routed via supervisor RPC into the facet's SQL surface. - `get_sync_errors(sync_id, limit?)` — paginated rows from `sync_errors`, newest first, by `(sync_id, occurred_at DESC)`. Returns `{occurred_at, error_message, stack_trace, payload_preview, handler_version}` per row. Used by the sync agent during the probe loop (verify phase-2 signature checks) and by `update_sync` debugging. @@ -999,7 +1043,10 @@ SyncAgent: run_in_sync { entries.push(...res.entries); cursor = res.cursor; } - await facet.__write('tree.json', JSON.stringify({ _cursor: cursor, _updated_at: new Date().toISOString(), entries })); + // NDJSON layout (§3.4): one entry per line in tree.ndjson; cursor in meta.json. + const ndjson = entries.map(e => JSON.stringify(e)).join('\\n'); + await facet.__write('tree.ndjson', ndjson); + await facet.__write('meta.json', JSON.stringify({ _cursor: cursor, _updated_at: new Date().toISOString(), entry_count: entries.length })); return { entries: entries.length }; }" } @@ -1008,14 +1055,15 @@ Kernel DO: load exec bundle via env.LOADER.load(buildExecWorkerCode({ code → calls FacetBridge.runInSync(facet, fetch) → exec returns { ok: true, result: { entries: 1247 }, logs: [...] } → tool returns the result to the agent -SyncAgent: (sees {entries: 1247}; tree.json is now populated) -SyncAgent: list_mirror sync_id → ["tree.json"] -SyncAgent: read_mirror sync_id, "tree.json" → confirms shape +SyncAgent: (sees {entries: 1247}; tree.ndjson + meta.json are now populated) +SyncAgent: list_mirror sync_id → ["meta.json", "tree.ndjson"] +SyncAgent: read_mirror sync_id, "meta.json" → { _cursor: "AAH9p3...", _updated_at: "...", entry_count: 1247 } +SyncAgent: read_mirror sync_id, "tree.ndjson" → first ~64KB; spot-checks first 50 lines parse as the expected entry shape SyncAgent: get_sync_errors sync_id → empty SyncAgent: streamText finishes (model emits stop signal — no further tools called) ↓ Consumer: runSyncAgent returns { ok: true, flavor: "sync-engine" }; queue message ack. - (Nothing else persists. The skill md + sync_engine row + facet + tree.json + registered webhook ARE the deliverable.) + (Nothing else persists. The skill md + sync_engine row + facet + meta.json + tree.ndjson + registered webhook ARE the deliverable.) ↓ [next time the user types in any thread, dropbox is in , the agent reads skills/integrations/dropbox.md when relevant] ``` @@ -1030,17 +1078,21 @@ Kernel: const facet = getOrCreateFacet(this, row) // ctx.facets.get Kernel: await facet.verifySignature(body, headers, row.webhook_secret) → true Kernel: await facet.handleWebhook(body, headers) ↓ (inside facet wrapper → user handler) - const tree = JSON.parse(await ctx.mirror.read("tree.json")) // MirrorFS RPC to supervisor + const meta = JSON.parse(await ctx.mirror.read("meta.json")) // MirrorFS RPC: small file const r = await ctx.fetch("https://api.dropboxapi.com/2/files/list_folder/continue", { method: "POST", headers: { "X-Pd-App": "dropbox", "Content-Type": "application/json" }, - body: JSON.stringify({ cursor: tree._cursor }) - }) // HttpGateway RPC to supervisor → Pipedream → Dropbox + body: JSON.stringify({ cursor: meta._cursor }) + }) // HttpGateway RPC: supervisor → Pipedream → Dropbox const data = await r.json() - // apply data.entries (added/modified) and data.entries with .tag='deleted' to tree.entries - tree._cursor = data.cursor - tree._updated_at = new Date().toISOString() - await ctx.mirror.write("tree.json", JSON.stringify(tree)) // MirrorFS RPC to supervisor + const lines = (await ctx.mirror.read("tree.ndjson")).split("\n") // MirrorFS RPC: ndjson body + const byId = new Map(lines.filter(Boolean).map(l => { const e = JSON.parse(l); return [e.id, e] })) + for (const e of data.entries) { + if (e['.tag'] === 'deleted') byId.delete(e.id); else byId.set(e.id, e); + } + const nextNdjson = [...byId.values()].map(e => JSON.stringify(e)).join("\n") + await ctx.mirror.write("tree.ndjson", nextNdjson) // MirrorFS RPC: rewrite ndjson + await ctx.mirror.write("meta.json", JSON.stringify({ _cursor: data.cursor, _updated_at: new Date().toISOString(), entry_count: byId.size })) return new Response("ok", { status: 200 }) ↓ Kernel: UPDATE sync_engine SET last_synced_at @@ -1056,8 +1108,9 @@ Main agent's turn begins. Context block: includes dropbox. System prompt teaches "check skills/integrations/.md". Skill auto-load: skills/integrations/dropbox.md is in scope. Main agent decides: this is a dropbox structure question, mirror has the answer. -Main agent: read_file integrations/dropbox/tree.json (returns the full json — file tree, no contents) -Main agent: (inline grep for path starts with /Personal/Tax/) +Main agent: grep_file { pattern: "\"path\":\"/Personal/Tax/", path: "integrations/dropbox/tree.ndjson" } + → returns ~6 matching lines, one entry-line of JSON each. + (Tree could be 50k entries / 8MB; only the matching lines crossed into context.) Main agent: Replies with the list of files + brief summary. ``` @@ -1073,8 +1126,8 @@ Skill auto-load: skills/integrations/dropbox.md is in scope. Skill says: headers: { 'X-Pd-App': 'dropbox', 'Dropbox-API-Arg': JSON.stringify({path: '/path/to/file'}) } })" -Main agent: read_file integrations/dropbox/tree.json -Main agent: (inline grep for /Personal/Tax/ AND modified > 2025-01-01) +Main agent: grep_file { pattern: "\"path\":\"/Personal/Tax/", path: "integrations/dropbox/tree.ndjson" } +Main agent: (filters returned entries by modified > 2025-01-01 in reasoning) → 4 matching entries, with paths Main agent: (decides to summarize — needs contents) Main agent: exec_code with JS that: @@ -1138,6 +1191,25 @@ exec_code returns success. ## 6. Stacked PR slicing (recap) +The work splits into one independent PR for the main-agent read primitives plus three stacked PRs for the sync agent itself. PR-R1 stands alone — it changes main-agent tools only, has no dependency on sync work, and unblocks the design assumptions baked into PR-F1/F2/F3 (the 64KB read cap and `grep_file` are what make NDJSON mirrors worth writing). Land PR-R1 first or in parallel; PR-F1 onward depend on it for their smoke tests but not their compile path. + +### PR-R1 — Main-agent R2 read primitives (independent prerequisite) +**Files modified:** +- `apps/kernel/src/tools/read-file.ts` — extend with `offset?: number` (1-based line) + `limit?: number` (default 2000) params; add 64KB soft cap on full reads; return text with the same line-number gutter (`\t`) Claude Code uses so the agent can address sub-ranges in follow-up calls. Above the cap with no offset/limit: error `file is N bytes; use grep_file or call read_file with {offset, limit}`. Stays `wrappedTool({ touchesFS: false })`. + +**Files created:** +- `apps/kernel/src/tools/grep-file.ts` — new tool; `{ pattern, path?, prefix?, glob?, regex_flags?, context_lines?, max_matches? }`. Compiles `pattern` via `new RegExp(pattern, regex_flags ?? "m")`. Scope: exactly one of `path` (single file) OR `prefix` + optional `glob` (recurses, filters via micromatch or equivalent). Reads each matched file from R2, scans line-by-line, collects up to `max_matches` (default 100) `{ file, line, match, before: string[], after: string[] }` records. Truncated runs return `{ matches, truncated: true }`. `wrappedTool({ touchesFS: false })`. + +**Files modified (wiring):** +- `apps/kernel/src/index.ts` — register `grep_file` alongside `read_file` in the main-agent tool surface. +- main-agent system prompt — one-line addition teaching the cap and the grep_file fallback. (Probably 2 sentences: "Files over ~64KB return an error; use grep_file for substring/regex search across files, read_file with {offset, limit} for known slices.") + +**Smoke:** +1. Seed R2 with a 200KB JSON file; `read_file` returns the cap error pointing at grep_file. +2. Same file with `{offset: 1, limit: 50}` returns the first 50 lines with a line-number gutter. +3. Seed an NDJSON file with 5000 entries; `grep_file({ pattern: "specific-substring", path })` returns ≤100 matching lines, each with 2 lines of context, all under a few KB. +4. `grep_file({ pattern: "...", prefix: "skills/integrations/", glob: "*.md" })` matches across all skill files. Truncates at 100 matches with `truncated: true`. + ### PR-F1 — Foundation primitives (no agent, no queue consumer yet) **Files created:** - `packages/models/src/schema/sync-engine.ts` — `sync_engine` + `sync_errors` drizzle tables (full schema as in §3.3) @@ -1156,7 +1228,7 @@ exec_code returns success. - `apps/kernel/src/agent/system-prompt.ts` — add the convention sentence: "For any integration in ``, `skills/integrations/.md` (if present) contains usage guidance — read it before invoking that integration." No new injected block. - `wrangler.jsonc` — declare `HttpGateway` and `MirrorFS` as named WorkerEntrypoints if not auto-discovered -**Smoke:** install fixture handler via a dev-only RPC that simulates `create_sync_engine` → fixture's `registerWebhook` runs against real Dropbox (Pipedream-proxied) → change a file in Dropbox → empty ping arrives → facet handles → `integrations/dropbox/tree.json` updates within 30s. No agent and no queue consumer involved. +**Smoke:** install fixture handler via a dev-only RPC that simulates `create_sync_engine` → fixture's `registerWebhook` runs against real Dropbox (Pipedream-proxied) → change a file in Dropbox → empty ping arrives → facet handles → `integrations/dropbox-fixture/tree.ndjson` updates within 30s. No agent and no queue consumer involved. (Fixture uses the same NDJSON + meta.json layout the agent will produce in PR-F2 — simpler to share R2-FS assertion helpers across both PRs.) ### PR-F2 — Queue trigger + sync agent + Dropbox e2e **Files created:** @@ -1220,7 +1292,7 @@ exec_code returns success. 4. Dev-only RPC installs the fixture dropbox handler into `sync_engine` with status='active', strategy='webhook_channel', mirror_prefix='integrations/dropbox-fixture/'. 5. Fixture's `registerWebhook` runs successfully against real Dropbox API (Pipedream-proxied, against a test account). 6. Webhook URL `https:///webhook/dropbox/` returns 200 to a signed Dropbox POST. -7. Touch a file in Dropbox → empty ping arrives → handler reads cursor from `tree.json`, calls list_folder/continue, applies delta → `integrations/dropbox-fixture/tree.json` reflects the change within 30s. +7. Touch a file in Dropbox → empty ping arrives → handler reads cursor from `meta.json`, calls list_folder/continue, applies delta to `tree.ndjson` lines, writes both files back → mirror reflects the change within 30s. 8. Bad-signature request returns 401, `sync_errors` row recorded with `error_message="signature verification rejected"`. 9. Main agent's system prompt has the convention sentence pointing at `skills/integrations/.md`; with the fixture installed, `/sync status` confirms the fixture row exists. 10. `/sync delete dropbox` removes the row, list-and-deletes every key under `integrations/dropbox-fixture/`, fixture's unregisterWebhook is called. @@ -1230,7 +1302,7 @@ exec_code returns success. 2. Queue receives the message; consumer fires; `runSyncAgent` begins (visible in `wrangler tail` as the consumer's `console.log` output streams tool calls). 3. Sync agent completes within ~5–10 min (target; cost ceiling enforced). 4. `skills/integrations/dropbox.md` exists in R2 and accurately describes the test account's folder structure in prose. -5. `integrations/dropbox/tree.json` (or whatever filename the agent chose — likely tree.json given §3.4 example) exists, has the expected shape, includes top-level folders. `list_mirror` over the prefix returns the files the skill's "Mirror layout" section claims will be there. +5. `integrations/dropbox/tree.ndjson` (and `meta.json`, or whatever layout the agent chose — likely the §3.4 NDJSON shape) exists, has the expected line-oriented shape, includes top-level folders. `list_mirror` over the prefix returns the files the skill's "Mirror layout" section claims will be there. `grep_file` against `tree.ndjson` for a known path returns the expected entry-line. 6. Main agent (next turn, in any user thread) sees `dropbox` in `` and reads `skills/integrations/dropbox.md` when relevant. 7. Ask main agent: "where are my tax filings?" → it reads the skill, greps the mirror, answers correctly with paths. 8. Ask main agent: "upload a one-pager for AcmeCorp under the right portfolio folder" → it picks the path from the skill, runs `exec_code`, file lands in Dropbox. @@ -1257,7 +1329,7 @@ exec_code returns success. | 3 | No `api_hosts` allowlist | Pipedream proxy is the egress gate; redundant in agent-os. | | 4 | No `ctx.agent` in handler runtime | Per user; inline JS handlers are enough; runtime LLM reasoning out of scope. | | 5 | Agent authors json shape | Sauna parity; "machine builds machine"; schema_doc carries the human-readable description. | -| 6 | Webhook/cursor state placement is the agent's call: facet SQLite row (preferred default for high-churn or multi-field state — Gmail's `webhook_config: {channel_id, expires_at, history_id}`, Drive's per-channel rows), mirror-embedded (`tree.json#_cursor` for Dropbox's single low-churn cursor), or sidecar mirror file. **The supervisor never reads cursors** — placement is purely a handler-internal performance decision. Sync agent system prompt (§3.10.3 Phase 4) teaches the rule of thumb: if only the handler reads it, prefer SQL; if the main agent reads it, mirror. The runtime imposes no defaults. | +| 6 | Webhook/cursor state placement is the agent's call: facet SQLite row (preferred default for high-churn or multi-field state — Gmail's `webhook_config: {channel_id, expires_at, history_id}`, Drive's per-channel rows), mirror-embedded in a small sidecar file (Dropbox's `meta.json` next to `tree.ndjson`), or any other shape that fits. **The supervisor never reads cursors** — placement is purely a handler-internal performance decision. Sync agent system prompt (§3.10.3 Phase 4) teaches the rule of thumb: if only the handler reads it, prefer SQL; if the main agent reads it, mirror; never embed a churning cursor in a multi-MB index file. The runtime imposes no defaults. | | 6b | Two schemas authored per Branch-A sync: `schema_doc` (prose, describes the R2 mirror layout — which files exist, what shape each has) + `schema_ddl` (SQL, defines facet SQLite tables) | Two distinct data layers serve different purposes. Mirror is structural index for main agent (R2 KV space at `integrations//`, grep-friendly per file). Facet SQLite is handler-internal state (dedup keys, raw_events for probe loop, multi-cursor state). The agent designs both at create_sync_engine time; supervisor runs DDL at facet bootstrap. | | 6c | Mirror is a prefix-scoped R2 key-value space, not a single file | Single-file forces whole-tree rewrites on every delta (expensive for high-write integrations like Gmail) and makes large mirrors hostile to grep-friendly `read_file`. Prefix layout lets the agent split by useful axis (label, repo, base, etc.); main agent finds the right file via the skill md's "Mirror layout" section. MirrorFS WorkerEntrypoint enforces the prefix scope; facet can't escape. | | 7 | Sync agent is NOT a tool — it runs in a Cloudflare Queue consumer triggered by connection event | Decouples connection detection from agent execution. Sync agent runs are minutes-long and burn tokens — running them inline on the connection-detection path would block other work and blow Worker timeouts. Queue gives retries, DLQ, back-pressure for free. | @@ -1270,3 +1342,5 @@ exec_code returns success. | 11 | 3-PR stack (not 4) | Foundation primitives + queue trigger + agent + dropbox e2e split as F1/F2; F3 covers the other archetypes (Linear stable, gcal channel, Notion poll-or-webhook-bus). | | 12 | Master spec + per-PR plans | One coherent architecture doc; per-PR plans land when each PR kicks off so they reflect the implementation reality. | | 13 | **Primitives stay neutral; the system prompt teaches behavior** (§1.2) | The supervisor exposes unopinionated read/write/list/delete + sql.exec + fetch primitives. It never imposes where cursors go, how the mirror is shaped, what tables to declare, or which surface (mirror vs SQL) is "correct" for a given piece of state. Best practices live in the sync agent's system prompt where the LLM can apply judgment. Adding defaults to the runtime would prematurely lock in choices that vary per integration; adding them to the prompt keeps the runtime flexible AND gives us a single place to update guidance as we learn what works. | +| 14 | **Main agent gets `read_file` + `grep_file`, NOT one composite tool** (§3.1.1) | Return shapes diverge (text vs match records), argument surfaces diverge (offset/limit vs pattern/glob/context), failure-recovery hints diverge. Folding both into one schema produces a union return type the agent has to discriminate on every call, and LLMs are measurably worse on union returns than focused ones. Claude Code keeps Read and Grep separate for the same reasons. Both are generic R2 FS tools (no mirror-specific framing) so they also serve skill files, logs, anything in R2. | +| 15 | **64KB soft cap on `read_file`; NDJSON or sharding preferred for high-cardinality mirrors** (§3.1.1, §3.4) | Whole-object reads of a multi-MB mirror would torch the main agent's context window — defeating the index-not-content premise entirely. The cap forces large reads through `grep_file` (server-side regex, only matches cross into context) or paginated `{offset, limit}` reads. NDJSON makes the "one entry per line" model `grep_file`-friendly by construction; sharding lets the agent pick the right small file before reading. Cap is a guardrail with a clear error+hint, not a hard refusal — explicit `{offset, limit}` bypasses it. | From 06f0b9d69724b214516e07717aab4f46d7397bd0 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 11:57:42 +0530 Subject: [PATCH 07/13] =?UTF-8?q?docs(plan):=20PR-R1=20read=20primitives?= =?UTF-8?q?=20=E2=80=94=20read=5Ffile=20slice/gutter=20+=20grep=5Ffile?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bite-sized plan covering: - read_file: add offset/limit (1-based line slicing), line-number gutter on output, touchesFS:false. Keeps existing 64KB truncate semantics for the no-args case. - grep_file: new server-side regex tool over R2 text files. path | prefix scope, regex_flags, context_lines, max_matches. Binary files skipped. - One-sentence main-agent system prompt update. - 9-step agent-driven smoke (no unit tests per memory). - PR raised against main, independent of the sync-agent spec PR. Decisions locked pre-plan: keep truncate semantics (not error-on-overflow), line gutter on output, regex string for grep pattern. Co-Authored-By: Claude Opus 4.7 (1M context) --- .../plans/2026-05-18-pr-r1-read-primitives.md | 605 ++++++++++++++++++ 1 file changed, 605 insertions(+) create mode 100644 docs/superpowers/plans/2026-05-18-pr-r1-read-primitives.md diff --git a/docs/superpowers/plans/2026-05-18-pr-r1-read-primitives.md b/docs/superpowers/plans/2026-05-18-pr-r1-read-primitives.md new file mode 100644 index 0000000..367cecb --- /dev/null +++ b/docs/superpowers/plans/2026-05-18-pr-r1-read-primitives.md @@ -0,0 +1,605 @@ +# PR-R1 — Main-agent R2 read primitives Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. +> +> **Model selection:** opus for all tasks (per memory `feedback_subagent_opus_default`). +> +> **Tests:** per memory `feedback_minimal_tests_in_plans`, no TDD steps. One agent-driven smoke at the end covers the matrix. + +**Goal:** Extend `read_file` with line-based pagination + a line-number gutter, and add a new `grep_file` tool that runs server-side regex over R2 files and returns only matching lines plus context. Together these stop the main agent from torching its context window on multi-MB mirror files. + +**Architecture:** Both tools live inside `buildFSTools` in `apps/kernel/src/tools/fs-tools.ts` (auto-picked up by `buildTools` and `buildSubAgentTools` in `tools/index.ts`). Both use the existing `EnvFS` surface via `buildKernelEnvFS(env, threadId, callId)` — no new R2 plumbing. Both are `wrappedTool({ touchesFS: false })` (read-only) so they skip the per-call worktree. + +**Tech Stack:** TypeScript / Cloudflare Workers / AI SDK v5 / Zod / existing `EnvFS` + `R2Bucket` bindings. + +**Spec reference:** `docs/superpowers/specs/2026-05-17-sync-agent-design.md` §3.1.1 ("Main-agent read primitives over R2") and §6 PR-R1. + +**Decisions confirmed pre-plan (2026-05-18):** +- 64KB cap on `read_file` keeps **truncate** semantics (existing behavior). New `offset`/`limit` is the opt-in path for explicit slices and bypasses the cap. +- `read_file` content gets a **line-number gutter** (`\t` per line) so the agent can address sub-ranges by number in follow-up calls. Return shape's `content` field changes accordingly. +- `grep_file` pattern is a **regex string** compiled with `new RegExp(pattern, regex_flags ?? "m")`. + +--- + +## House-keeping rules + +Apply to every task. The implementer must NOT generate any of the following inside source files: + +- **No PR-letter markers** — no `// NEW (PR-R1)`, no `// ─── PR-R1 ───`, etc. `git blame` already tracks provenance. +- **No "// NEW" / "// MODIFIED" / "// CHANGED" stickers.** +- **JSDoc explaining WHY a field exists is fine.** Comments explaining "this is new" are not. + +If a code sample in this plan accidentally carries such a marker, strip it before pasting. + +--- + +## File structure + +- **Modify** `apps/kernel/src/tools/fs-tools.ts` — extend `read_file`, add `grep_file`. Both inside `buildFSTools`. +- **Modify** `apps/kernel/src/agent/system-prompt.ts` — one-sentence addition to the FS-tools paragraph (around line 77) that names `grep_file` and the 64KB/offset-limit pattern. +- **No changes** to `apps/kernel/src/tools/index.ts` — `buildFSTools` is already wired into `buildTools` and `buildSubAgentTools`; `grep_file` rides along automatically. +- **No changes** to `apps/kernel/src/fs/env-fs.ts` or `path.ts` — both tools use existing primitives (`fs.stat`, `fs.readFile`, `fs.listFiles`). +- **No new files** in this PR. + +--- + +## Task 1: Extend `read_file` with offset/limit + line gutter + touchesFS:false + +**Files:** +- Modify: `apps/kernel/src/tools/fs-tools.ts` — the `readFile: wrappedTool({ name: "read_file", ... })` block (currently spans roughly lines 15–72). + +The current tool truncates content past 64KB and returns the raw decoded text. We keep that for the no-offset case but layer in (a) a line-number gutter on output, (b) explicit `offset` / `limit` line-based slicing that bypasses the cap, and (c) `touchesFS: false` so reads don't open a worktree. + +- [ ] **Step 1: Add a `formatWithLineGutter` helper at module top** + +Place near the existing constants (`TEXT_CT_RE`, `READ_CAP`, `WRITE_CAP`). + +```ts +/** + * Render text as `\t` per line so the agent can address + * sub-ranges by number in follow-up read_file / grep_file calls. + * Line numbers start at `startLine` (1-based; caller chooses). + */ +function formatWithLineGutter(text: string, startLine: number): string { + const lines = text.split("\n"); + // Trailing newline produces an empty final element — preserve it as a + // blank line so the rendered output round-trips with the file. + return lines + .map((line, i) => `${startLine + i}\t${line}`) + .join("\n"); +} +``` + +- [ ] **Step 2: Extend the `inputSchema` with optional `offset` and `limit`** + +Replace the existing `inputSchema: z.object({ path: ... })` with: + +```ts +inputSchema: z.object({ + path: z + .string() + .startsWith("r2://") + .describe( + "Full r2:// path, e.g. r2://memory/USER_PROFILE.md, r2://my-files/notes.csv, r2://artifacts/data.json" + ), + offset: z + .number() + .int() + .min(1) + .optional() + .describe( + "1-based line number to start reading from. Combined with `limit`, returns a contiguous line range. Bypasses the 64KB cap (use this to read deeper into big files like NDJSON indexes)." + ), + limit: z + .number() + .int() + .min(1) + .max(10000) + .optional() + .describe( + "Maximum number of lines to return when `offset` is set. Defaults to 2000." + ) +}) +``` + +- [ ] **Step 3: Update the description to teach the new contract** + +Replace the current `description` string for `read_file` with: + +```ts +description: + "Read a text file from R2 with a line-number gutter (`\\t` per line) so you can address sub-ranges in follow-up calls. Default behavior (no offset/limit): returns the first 64KB with `truncated:true` + full `size` if the file is bigger — switch to `grep_file` for substring/regex search or call again with {offset, limit} for a deeper slice. With `offset` (1-based line) + optional `limit` (default 2000 lines), returns the explicit line range and bypasses the 64KB cap. Memory files, CSV/JSON/NDJSON/YAML/XML/HTML/code, markdown, logs, configs all work; refuses binary content (PDFs, images, archives — use process_attachment or env.FS.readFile inside exec_code).", +``` + +- [ ] **Step 4: Add `touchesFS: false` to the tool def** + +In the `readFile: wrappedTool({ ... })` block, insert `touchesFS: false,` next to `needsApproval: false,`. This is a small bug fix — read-only tools should not open a worktree per call (per memory `feedback_wrappedtool_default_readonly`). + +- [ ] **Step 5: Rewrite the `execute` body to handle three paths** + +Replace the existing `execute` function for `read_file` with: + +```ts +execute: async ({ path, offset, limit }, ctx) => { + const fs = buildKernelEnvFS(ctx.env, ctx.threadId, ctx.callId); + try { + const meta = await fs.stat(path); + if (!meta) { + return { ok: false as const, error: "not_found" as const, path }; + } + if (!TEXT_CT_RE.test(meta.contentType)) { + return { + ok: false as const, + error: "binary_not_supported" as const, + detail: `contentType ${meta.contentType} is binary; use process_attachment(path, prompt) or env.FS.readFile in exec_code`, + path, + contentType: meta.contentType, + size: meta.size + }; + } + const { content, contentType, size } = await fs.readFile(path); + const text = new TextDecoder().decode(content); + + // Explicit-slice path: offset/limit set. Bypass the 64KB cap; the + // caller asked for a specific line range on purpose. + if (offset !== undefined) { + const effLimit = limit ?? 2000; + const lines = text.split("\n"); + const startIdx = offset - 1; + if (startIdx >= lines.length) { + return { + ok: true as const, + path, + content: "", + contentType, + size, + startLine: offset, + endLine: offset - 1, + totalLines: lines.length, + truncated: false as const + }; + } + const endIdx = Math.min(startIdx + effLimit, lines.length); + const slice = lines.slice(startIdx, endIdx).join("\n"); + return { + ok: true as const, + path, + content: formatWithLineGutter(slice, offset), + contentType, + size, + startLine: offset, + endLine: startIdx + (endIdx - startIdx), + totalLines: lines.length, + truncated: false as const + }; + } + + // Default path: no offset/limit. Apply 64KB cap, truncate with flag. + if (text.length > READ_CAP) { + const head = text.slice(0, READ_CAP); + return { + ok: true as const, + path, + content: formatWithLineGutter(head, 1), + contentType, + size, + truncated: true as const + }; + } + return { + ok: true as const, + path, + content: formatWithLineGutter(text, 1), + contentType, + size, + truncated: false as const + }; + } catch (e) { + return mapEnvFSError("read_file", e); + } +} +``` + +- [ ] **Step 6: Commit** + +```bash +git add apps/kernel/src/tools/fs-tools.ts +git commit -m "feat(tools): read_file gets offset/limit + line gutter; touchesFS=false" +``` + +--- + +## Task 2: Implement `grep_file` + +**Files:** +- Modify: `apps/kernel/src/tools/fs-tools.ts` — add a new entry to the object returned by `buildFSTools` (place it after `findFiles` and before `deleteFile` so read tools cluster). + +Server-side regex over R2 files. Scope by single `path` OR by `prefix` (recurses). Returns matching lines with surrounding context — only matches cross into the agent's context window, never the whole file. + +- [ ] **Step 1: Add a `scanFileForMatches` helper near the top of the file** + +Place after `formatWithLineGutter` from Task 1. + +```ts +interface GrepMatch { + file: string; + line: number; + match: string; + before: string[]; + after: string[]; +} + +/** + * Scan a single file's text for regex matches. Returns up to `remaining` + * matches, each with `contextLines` lines of before/after context. + * + * The regex is reset between line scans (no global-flag state leakage). + */ +function scanFileForMatches( + filePath: string, + text: string, + regex: RegExp, + contextLines: number, + remaining: number +): GrepMatch[] { + if (remaining <= 0) return []; + const lines = text.split("\n"); + const out: GrepMatch[] = []; + for (let i = 0; i < lines.length && out.length < remaining; i++) { + // `lastIndex` reset guards against /g flag accumulating state. + regex.lastIndex = 0; + if (!regex.test(lines[i]!)) continue; + const before = lines.slice(Math.max(0, i - contextLines), i); + const after = lines.slice(i + 1, Math.min(lines.length, i + 1 + contextLines)); + out.push({ + file: filePath, + line: i + 1, // 1-based to match read_file's gutter + match: lines[i]!, + before, + after + }); + } + return out; +} +``` + +- [ ] **Step 2: Add the `grepFile` tool to `buildFSTools`'s returned object** + +Insert this block between `findFiles: wrappedTool({...}),` and `deleteFile: wrappedTool({...}),`. Keep return-object property order: read tools first, then write tools. + +```ts +grepFile: wrappedTool( + { + name: "grep_file", + description: + "Server-side regex search over R2 text files. Returns only matching lines + N lines of context — the file itself stays in the Worker, so your context budget tracks matches, not file size. Use this instead of read_file whenever a mirror file might exceed 64KB or you need to find by predicate across many files. Scope: exactly one of `path` (single file) or `prefix` (recurses, reads every text file under the prefix). `pattern` is a JS regex string; default flags are `m` (line-anchored). Caps: 100 matches across all files in scope by default, set `max_matches` to raise (max 1000). Binary files are skipped automatically.", + inputSchema: z.object({ + pattern: z + .string() + .min(1) + .describe( + "JS regex string (no slashes). Examples: '\"path\":\"/Personal/Tax/' (substring), '^id:[a-f0-9]+\\\\b' (anchored), 'TODO|FIXME' (alternation)." + ), + path: z + .string() + .startsWith("r2://") + .optional() + .describe( + "Full r2:// path of a single file to search. Mutually exclusive with `prefix`." + ), + prefix: z + .string() + .startsWith("r2://") + .optional() + .describe( + "r2:// prefix to recurse under (e.g. 'r2://skills/' or 'r2://integrations/dropbox/'). Mutually exclusive with `path`. Reads every text file at that prefix and runs the pattern across each." + ), + regex_flags: z + .string() + .regex(/^[gimsuy]*$/) + .optional() + .describe( + "Regex flags. Default 'm' (line-anchored). Pass 'mi' for case-insensitive, 'ms' for dotall, etc. Avoid 'g' — it's already implicit per-line." + ), + context_lines: z + .number() + .int() + .min(0) + .max(10) + .optional() + .describe( + "Lines of context before/after each match. Default 2." + ), + max_matches: z + .number() + .int() + .min(1) + .max(1000) + .optional() + .describe( + "Cap on total matches returned across all files in scope. Default 100. Past the cap the result carries truncated:true so you know to narrow." + ) + }), + needsApproval: false, + touchesFS: false, + execute: async ( + { pattern, path, prefix, regex_flags, context_lines, max_matches }, + ctx + ) => { + // Exactly-one-of scope validation. + if ((path === undefined) === (prefix === undefined)) { + return { + ok: false as const, + error: "bad_scope" as const, + detail: + "specify exactly one of `path` (single file) or `prefix` (recursive search)" + }; + } + + // Compile the regex up front so a bad pattern surfaces before any R2 read. + let regex: RegExp; + try { + regex = new RegExp(pattern, regex_flags ?? "m"); + } catch (e) { + return { + ok: false as const, + error: "bad_pattern" as const, + detail: e instanceof Error ? e.message : String(e) + }; + } + + const ctxLines = context_lines ?? 2; + const cap = max_matches ?? 100; + const fs = buildKernelEnvFS(ctx.env, ctx.threadId, ctx.callId); + const matches: GrepMatch[] = []; + + try { + if (path !== undefined) { + const meta = await fs.stat(path); + if (!meta) { + return { ok: false as const, error: "not_found" as const, path }; + } + if (!TEXT_CT_RE.test(meta.contentType)) { + return { + ok: false as const, + error: "binary_not_supported" as const, + detail: `contentType ${meta.contentType} is binary; grep_file is text-only`, + path, + contentType: meta.contentType + }; + } + const { content } = await fs.readFile(path); + const text = new TextDecoder().decode(content); + matches.push( + ...scanFileForMatches(path, text, regex, ctxLines, cap - matches.length) + ); + } else { + // prefix scope — enumerate files, skip binaries, scan each. + const files = await fs.listFiles(prefix!); + for (const f of files) { + if (matches.length >= cap) break; + if (!TEXT_CT_RE.test(f.contentType)) continue; + const { content } = await fs.readFile(f.path); + const text = new TextDecoder().decode(content); + matches.push( + ...scanFileForMatches(f.path, text, regex, ctxLines, cap - matches.length) + ); + } + } + } catch (e) { + return mapEnvFSError("grep_file", e); + } + + return { + ok: true as const, + matches, + truncated: matches.length >= cap + }; + } + }, + perTurn +), +``` + +- [ ] **Step 3: Commit** + +```bash +git add apps/kernel/src/tools/fs-tools.ts +git commit -m "feat(tools): add grep_file for server-side regex over R2" +``` + +--- + +## Task 3: Update the main-agent system prompt + +**Files:** +- Modify: `apps/kernel/src/agent/system-prompt.ts` — the `` block's FS-tools sentence (currently around line 77). + +The convention line about `skills/integrations/.md` is **PR-F1's** job, not R1's. R1's only prompt change is naming `grep_file` and teaching the cap pattern in the existing FS-tools paragraph. + +- [ ] **Step 1: Locate the FS-tools sentence** + +In `system-prompt.ts`, find the line that begins: + +``` +**read_file / write_file / edit_file / list_files / find_files / move_file / delete_file / get_signed_url** for direct R2 ops on text content. +``` + +- [ ] **Step 2: Replace with the extended version** + +Use Edit tool with `old_string` = the full existing sentence + its trailing paragraph (up to the blank line before the next `**...**` capability). `new_string`: + +``` +**read_file / grep_file / write_file / edit_file / list_files / find_files / move_file / delete_file / get_signed_url** for direct R2 ops on text content. \`grep_file\` runs regex server-side and returns only matching lines + context — reach for it whenever the file might exceed 64KB or you need to find by predicate across many files. \`read_file\` returns a line-number gutter; if you get \`truncated:true\` on a big file, switch to \`grep_file\` or call \`read_file\` again with \`{offset, limit}\` for a deeper slice. \`edit_file\` is the surgical one: read-then-replace a unique anchor, no full-file rewrite. Use it for memory updates and any single-section edit. If your work around the FS call is more than three lines of logic, switch to \`env.FS\` inside exec_code instead. +``` + +(The backtick-escaping is for the template-literal context — keep the literal backticks in the file since `SYSTEM_PROMPT` is a backtick-quoted string.) + +- [ ] **Step 3: Commit** + +```bash +git add apps/kernel/src/agent/system-prompt.ts +git commit -m "feat(prompt): teach main agent grep_file + read_file slice pattern" +``` + +--- + +## Task 4: Type-check and lint + +- [ ] **Step 1: Run the typechecker on the kernel package** + +```bash +pnpm --filter @agent-os/kernel typecheck +``` + +Expected: clean, no errors. If `wrappedTool`'s inferred return type complains about the new return-shape union members on `read_file` (added `startLine` / `endLine` / `totalLines` in some branches but not others), normalize by always returning those fields with sensible defaults — or accept the union and let TS narrow naturally; depends on what the typechecker actually says. + +- [ ] **Step 2: Run oxlint + oxfmt** + +Per memory `feedback_lint_format`, agent-os uses oxc tooling (not Prettier/ESLint). + +```bash +pnpm oxlint apps/kernel/src/tools/fs-tools.ts apps/kernel/src/agent/system-prompt.ts +pnpm oxfmt apps/kernel/src/tools/fs-tools.ts apps/kernel/src/agent/system-prompt.ts +``` + +Expected: no findings; oxfmt may rewrite formatting in place — re-stage if it does. + +- [ ] **Step 3: Commit any oxfmt-driven changes** + +If oxfmt rewrote anything: + +```bash +git add apps/kernel/src/tools/fs-tools.ts apps/kernel/src/agent/system-prompt.ts +git commit -m "style: oxfmt" +``` + +If nothing changed, skip. + +--- + +## Task 5: Agent-driven smoke (the actual verification) + +Per memory `feedback_minimal_tests_in_plans`, we don't write unit tests for tool features. Instead, an end-to-end smoke through the agent surface covers the matrix more authentically. + +**Prerequisites:** local kernel running (`pnpm dev` or whatever the current dev entrypoint is). Have one small (<64KB) text file and one large (>64KB) text file already seeded under any allowed R2 prefix — `r2://my-files/` is the simplest. If none exist, the smoke includes a write step to create them. + +- [ ] **Step 1: Seed test fixtures via the CLI (or skip if files already exist)** + +In the running CLI, ask the agent: +> "Write a small file at `r2://my-files/r1-smoke-small.md` with the markdown `# Hello\nworld` and a large file at `r2://my-files/r1-smoke-large.ndjson` with 5000 lines of NDJSON like `{\"id\":\"id-N\",\"path\":\"/folder/file-N.txt\"}` where N is the 1-indexed line number." + +Wait for both write_file confirmations. Verify the large file's size via `list_files r2://my-files/` — should be ~200KB+. + +- [ ] **Step 2: Verify `read_file` truncates large files and shows the gutter** + +Ask the agent: +> "Read `r2://my-files/r1-smoke-large.ndjson`." + +Expected agent output (paraphrased): notes the file is large, gets `truncated:true` with `size: ~200000+`. The content block in the tool result has `\t` per line. Confirm by asking the agent: "What was the line number prefix format? Was the read truncated?" + +- [ ] **Step 3: Verify `read_file` with offset/limit returns a specific slice** + +Ask the agent: +> "Read lines 2500–2510 of that NDJSON file." + +Expected: agent calls `read_file` with `offset: 2500, limit: 11` (or similar). Tool result contains lines 2500-2510 only, prefixed with their line numbers, `truncated: false`, `totalLines: 5000`. Agent reports the entry IDs back correctly. + +- [ ] **Step 4: Verify `grep_file` on a single file** + +Ask the agent: +> "Find the entry with id `id-4242` in that NDJSON file using grep_file." + +Expected: agent calls `grep_file` with `pattern: "\"id\":\"id-4242\""` and the explicit `path`. Result has exactly one match with `line: 4242`, the matching JSON, and 2 lines of context before/after. Agent reports the path field correctly (`/folder/file-4242.txt`). + +- [ ] **Step 5: Verify `grep_file` on a prefix scope** + +Ask the agent: +> "Find every line that says 'Hello' across `r2://my-files/`." + +Expected: agent calls `grep_file` with `pattern: "Hello"` and `prefix: "r2://my-files/"`. Result includes the match from the small file. NDJSON file has none. `truncated: false`. + +- [ ] **Step 6: Verify `grep_file` cap + truncation flag** + +Ask the agent: +> "Find every line that contains 'id-' in that NDJSON file. Cap matches at 50." + +Expected: agent calls `grep_file` with `pattern: "id-", path: "r2://my-files/r1-smoke-large.ndjson", max_matches: 50`. Result has 50 matches and `truncated: true`. Agent recognizes the cap and either narrows or proceeds. + +- [ ] **Step 7: Verify bad-pattern error path** + +Ask the agent: +> "Run grep_file with the pattern `[unclosed` against any file." + +Expected: agent calls grep_file, gets `{ok: false, error: "bad_pattern", detail: "..."}`. Agent does NOT retry blindly; explains the regex error to the user. + +- [ ] **Step 8: Verify binary-file refusal** + +If a PDF or image exists under any prefix, ask: +> "Grep that pdf for 'foo'." + +Expected: `{ok: false, error: "binary_not_supported", contentType: "application/pdf"}`. Agent reports the file is binary. + +(If no binary file is handy, skip — covered by the contentType filter logic; not worth seeding a binary just for one path.) + +- [ ] **Step 9: Cleanup** + +Ask the agent: +> "Delete `r2://my-files/r1-smoke-small.md` and `r2://my-files/r1-smoke-large.ndjson`." + +Expected: two delete_file calls, both succeed. + +--- + +## Task 6: Raise the PR + +- [ ] **Step 1: Push the branch** + +```bash +git push -u origin feat/sync-r1-read-primitives +``` + +- [ ] **Step 2: Open the PR via gh** + +Base the PR on `main` (per memory `feedback_main_branch_workflow`). The spec lives on `feat/sync-agent-design` (open as PR #19). PR-R1 stands alone — it doesn't depend on the spec PR merging first, so basing on `main` keeps the stack simple and lets reviewers merge in any order. + +```bash +gh pr create --base main --title "feat(tools): read_file slice/gutter + new grep_file (PR-R1)" --body "$(cat <<'EOF' +## Summary +- Extends `read_file` with `offset`/`limit` (1-based line slicing) plus a line-number gutter on output. Default no-args behavior unchanged except for the gutter (still truncates at 64KB with `truncated:true`). +- Adds `grep_file`: server-side regex over R2 text files. Scope by single `path` or recursive `prefix`. Returns matching lines + N context lines. Caps at 100 matches by default. Binary files skipped automatically. +- Sets `touchesFS: false` on `read_file` (small bug fix — read-only tools shouldn't open a per-call worktree). +- One-sentence prompt update naming `grep_file` and teaching the `truncated:true → switch to grep` pattern. + +## Spec reference +- `docs/superpowers/specs/2026-05-17-sync-agent-design.md` §3.1.1 + §6 PR-R1 +- Independent prerequisite for the sync-agent stack (F1/F2/F3) — unblocks NDJSON-mirror design assumptions but has no compile-time dependency on them. + +## Test plan +- [ ] Seed a small md + a 5000-line NDJSON in `r2://my-files/` +- [ ] `read_file` on the NDJSON returns 64KB truncated head with line gutter + size +- [ ] `read_file` with `{offset: 2500, limit: 11}` returns the explicit slice, no truncation +- [ ] `grep_file` on `path` for a single id returns exactly one match with context +- [ ] `grep_file` on `prefix` matches across the small file +- [ ] `grep_file` with `max_matches: 50` returns 50 matches + `truncated: true` +- [ ] `grep_file` with `pattern: "[unclosed"` returns `{ok:false, error:"bad_pattern"}` +- [ ] Binary file → `{ok:false, error:"binary_not_supported"}` +- [ ] Cleanup deletes succeed + +🤖 Generated with [Claude Code](https://claude.com/claude-code) +EOF +)" +``` + +- [ ] **Step 3: Return the PR URL** + +Echo the URL so it's visible at the end of the run. + +--- + +## Done criteria + +- All checkboxes above are checked. +- The 9-step agent smoke (Task 5) passes end-to-end through the CLI. +- `typecheck` is clean, oxlint/oxfmt are quiet. +- PR is opened against `main` with the test-plan checklist visible. From 92195bd767ba29853ed8e1b2af023adb72f33064 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 12:12:13 +0530 Subject: [PATCH 08/13] feat(tools): read_file gets offset/limit + line gutter; touchesFS=false --- apps/kernel/src/tools/fs-tools.ts | 77 +++++++++++++++++++++++++++++-- 1 file changed, 73 insertions(+), 4 deletions(-) diff --git a/apps/kernel/src/tools/fs-tools.ts b/apps/kernel/src/tools/fs-tools.ts index 01f9635..6cfa1d3 100644 --- a/apps/kernel/src/tools/fs-tools.ts +++ b/apps/kernel/src/tools/fs-tools.ts @@ -10,23 +10,55 @@ const TEXT_CT_RE = const READ_CAP = 64 * 1024; const WRITE_CAP = 1024 * 1024; +/** + * Render text as `\t` per line so the agent can address + * sub-ranges by number in follow-up read_file / grep_file calls. + * Line numbers start at `startLine` (1-based; caller chooses). + */ +function formatWithLineGutter(text: string, startLine: number): string { + const lines = text.split("\n"); + // Trailing newline produces an empty final element — preserve it as a + // blank line so the rendered output round-trips with the file. + return lines + .map((line, i) => `${startLine + i}\t${line}`) + .join("\n"); +} + export const buildFSTools = (perTurn: PerTurnContext) => { return { readFile: wrappedTool( { name: "read_file", description: - "Read a text file from R2: memory files, CSV/JSON/YAML/XML/HTML/code, markdown, logs, configs. Refuses binary content (PDFs, images, archives); for those, process_attachment understands them, or env.FS.readFile inside exec_code gets you the raw bytes. Truncates at 64KB with truncated:true; the full size comes back so you can switch to exec_code (no cap there) when needed.", + "Read a text file from R2 with a line-number gutter (`\\t` per line) so you can address sub-ranges in follow-up calls. Default behavior (no offset/limit): returns the first 64KB with `truncated:true` + full `size` if the file is bigger — switch to `grep_file` for substring/regex search or call again with {offset, limit} for a deeper slice. With `offset` (1-based line) + optional `limit` (default 2000 lines), returns the explicit line range and bypasses the 64KB cap. Memory files, CSV/JSON/NDJSON/YAML/XML/HTML/code, markdown, logs, configs all work; refuses binary content (PDFs, images, archives — use process_attachment or env.FS.readFile inside exec_code).", inputSchema: z.object({ path: z .string() .startsWith("r2://") .describe( "Full r2:// path, e.g. r2://memory/USER_PROFILE.md, r2://my-files/notes.csv, r2://artifacts/data.json" + ), + offset: z + .number() + .int() + .min(1) + .optional() + .describe( + "1-based line number to start reading from. Combined with `limit`, returns a contiguous line range. Bypasses the 64KB cap (use this to read deeper into big files like NDJSON indexes)." + ), + limit: z + .number() + .int() + .min(1) + .max(10000) + .optional() + .describe( + "Maximum number of lines to return when `offset` is set. Defaults to 2000." ) }), needsApproval: false, - execute: async ({ path }, ctx) => { + touchesFS: false, + execute: async ({ path, offset, limit }, ctx) => { const fs = buildKernelEnvFS(ctx.env, ctx.threadId, ctx.callId); try { const meta = await fs.stat(path); @@ -45,11 +77,48 @@ export const buildFSTools = (perTurn: PerTurnContext) => { } const { content, contentType, size } = await fs.readFile(path); const text = new TextDecoder().decode(content); + + // Explicit-slice path: offset/limit set. Bypass the 64KB cap; the + // caller asked for a specific line range on purpose. + if (offset !== undefined) { + const effLimit = limit ?? 2000; + const lines = text.split("\n"); + const startIdx = offset - 1; + if (startIdx >= lines.length) { + return { + ok: true as const, + path, + content: "", + contentType, + size, + startLine: offset, + endLine: offset - 1, + totalLines: lines.length, + truncated: false as const + }; + } + const endIdx = Math.min(startIdx + effLimit, lines.length); + const slice = lines.slice(startIdx, endIdx).join("\n"); + return { + ok: true as const, + path, + content: formatWithLineGutter(slice, offset), + contentType, + size, + startLine: offset, + endLine: startIdx + (endIdx - startIdx), + totalLines: lines.length, + truncated: false as const + }; + } + + // Default path: no offset/limit. Apply 64KB cap, truncate with flag. if (text.length > READ_CAP) { + const head = text.slice(0, READ_CAP); return { ok: true as const, path, - content: text.slice(0, READ_CAP), + content: formatWithLineGutter(head, 1), contentType, size, truncated: true as const @@ -58,7 +127,7 @@ export const buildFSTools = (perTurn: PerTurnContext) => { return { ok: true as const, path, - content: text, + content: formatWithLineGutter(text, 1), contentType, size, truncated: false as const From e216615bb80fe36b37112b2ac0b7db651cb536da Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 12:12:50 +0530 Subject: [PATCH 09/13] feat(tools): add grep_file for server-side regex over R2 --- apps/kernel/src/tools/fs-tools.ts | 173 ++++++++++++++++++++++++++++++ 1 file changed, 173 insertions(+) diff --git a/apps/kernel/src/tools/fs-tools.ts b/apps/kernel/src/tools/fs-tools.ts index 6cfa1d3..f9dcf97 100644 --- a/apps/kernel/src/tools/fs-tools.ts +++ b/apps/kernel/src/tools/fs-tools.ts @@ -24,6 +24,47 @@ function formatWithLineGutter(text: string, startLine: number): string { .join("\n"); } +interface GrepMatch { + file: string; + line: number; + match: string; + before: string[]; + after: string[]; +} + +/** + * Scan a single file's text for regex matches. Returns up to `remaining` + * matches, each with `contextLines` lines of before/after context. + * + * The regex is reset between line scans (no global-flag state leakage). + */ +function scanFileForMatches( + filePath: string, + text: string, + regex: RegExp, + contextLines: number, + remaining: number +): GrepMatch[] { + if (remaining <= 0) return []; + const lines = text.split("\n"); + const out: GrepMatch[] = []; + for (let i = 0; i < lines.length && out.length < remaining; i++) { + // `lastIndex` reset guards against /g flag accumulating state. + regex.lastIndex = 0; + if (!regex.test(lines[i]!)) continue; + const before = lines.slice(Math.max(0, i - contextLines), i); + const after = lines.slice(i + 1, Math.min(lines.length, i + 1 + contextLines)); + out.push({ + file: filePath, + line: i + 1, // 1-based to match read_file's gutter + match: lines[i]!, + before, + after + }); + } + return out; +} + export const buildFSTools = (perTurn: PerTurnContext) => { return { readFile: wrappedTool( @@ -353,6 +394,138 @@ export const buildFSTools = (perTurn: PerTurnContext) => { perTurn ), + grepFile: wrappedTool( + { + name: "grep_file", + description: + "Server-side regex search over R2 text files. Returns only matching lines + N lines of context — the file itself stays in the Worker, so your context budget tracks matches, not file size. Use this instead of read_file whenever a mirror file might exceed 64KB or you need to find by predicate across many files. Scope: exactly one of `path` (single file) or `prefix` (recurses, reads every text file under the prefix). `pattern` is a JS regex string; default flags are `m` (line-anchored). Caps: 100 matches across all files in scope by default, set `max_matches` to raise (max 1000). Binary files are skipped automatically.", + inputSchema: z.object({ + pattern: z + .string() + .min(1) + .describe( + "JS regex string (no slashes). Examples: '\"path\":\"/Personal/Tax/' (substring), '^id:[a-f0-9]+\\\\b' (anchored), 'TODO|FIXME' (alternation)." + ), + path: z + .string() + .startsWith("r2://") + .optional() + .describe( + "Full r2:// path of a single file to search. Mutually exclusive with `prefix`." + ), + prefix: z + .string() + .startsWith("r2://") + .optional() + .describe( + "r2:// prefix to recurse under (e.g. 'r2://skills/' or 'r2://integrations/dropbox/'). Mutually exclusive with `path`. Reads every text file at that prefix and runs the pattern across each." + ), + regex_flags: z + .string() + .regex(/^[gimsuy]*$/) + .optional() + .describe( + "Regex flags. Default 'm' (line-anchored). Pass 'mi' for case-insensitive, 'ms' for dotall, etc. Avoid 'g' — it's already implicit per-line." + ), + context_lines: z + .number() + .int() + .min(0) + .max(10) + .optional() + .describe( + "Lines of context before/after each match. Default 2." + ), + max_matches: z + .number() + .int() + .min(1) + .max(1000) + .optional() + .describe( + "Cap on total matches returned across all files in scope. Default 100. Past the cap the result carries truncated:true so you know to narrow." + ) + }), + needsApproval: false, + touchesFS: false, + execute: async ( + { pattern, path, prefix, regex_flags, context_lines, max_matches }, + ctx + ) => { + // Exactly-one-of scope validation. + if ((path === undefined) === (prefix === undefined)) { + return { + ok: false as const, + error: "bad_scope" as const, + detail: + "specify exactly one of `path` (single file) or `prefix` (recursive search)" + }; + } + + // Compile the regex up front so a bad pattern surfaces before any R2 read. + let regex: RegExp; + try { + regex = new RegExp(pattern, regex_flags ?? "m"); + } catch (e) { + return { + ok: false as const, + error: "bad_pattern" as const, + detail: e instanceof Error ? e.message : String(e) + }; + } + + const ctxLines = context_lines ?? 2; + const cap = max_matches ?? 100; + const fs = buildKernelEnvFS(ctx.env, ctx.threadId, ctx.callId); + const matches: GrepMatch[] = []; + + try { + if (path !== undefined) { + const meta = await fs.stat(path); + if (!meta) { + return { ok: false as const, error: "not_found" as const, path }; + } + if (!TEXT_CT_RE.test(meta.contentType)) { + return { + ok: false as const, + error: "binary_not_supported" as const, + detail: `contentType ${meta.contentType} is binary; grep_file is text-only`, + path, + contentType: meta.contentType + }; + } + const { content } = await fs.readFile(path); + const text = new TextDecoder().decode(content); + matches.push( + ...scanFileForMatches(path, text, regex, ctxLines, cap - matches.length) + ); + } else { + // prefix scope — enumerate files, skip binaries, scan each. + const files = await fs.listFiles(prefix!); + for (const f of files) { + if (matches.length >= cap) break; + if (!TEXT_CT_RE.test(f.contentType)) continue; + const { content } = await fs.readFile(f.path); + const text = new TextDecoder().decode(content); + matches.push( + ...scanFileForMatches(f.path, text, regex, ctxLines, cap - matches.length) + ); + } + } + } catch (e) { + return mapEnvFSError("grep_file", e); + } + + return { + ok: true as const, + matches, + truncated: matches.length >= cap + }; + } + }, + perTurn + ), + deleteFile: wrappedTool( { name: "delete_file", From bb5a0321f1d4e1f81294d3918121af5e51e6d6b4 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 12:13:07 +0530 Subject: [PATCH 10/13] feat(prompt): teach main agent grep_file + read_file slice pattern --- apps/kernel/src/agent/system-prompt.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/kernel/src/agent/system-prompt.ts b/apps/kernel/src/agent/system-prompt.ts index 5772f3c..e5f8b99 100644 --- a/apps/kernel/src/agent/system-prompt.ts +++ b/apps/kernel/src/agent/system-prompt.ts @@ -74,7 +74,7 @@ You have five surfaces. Pick the one that matches the work, not the one that's m **computer_bash and the sandbox suite** for work that needs real Unix: pip, apt, ffmpeg, pandoc, git clones, multi-step builds, dev servers. Backed by an R2-mounted FUSE so \`/r2/*\` survives across turns. Slower than exec_code, much more capable. Use it when JS-only would be a stretch, not by default. -**read_file / write_file / edit_file / list_files / find_files / move_file / delete_file / get_signed_url** for direct R2 ops on text content. \`edit_file\` is the surgical one: read-then-replace a unique anchor, no full-file rewrite. Use it for memory updates and any single-section edit. If your work around the FS call is more than three lines of logic, switch to \`env.FS\` inside exec_code instead. +**read_file / grep_file / write_file / edit_file / list_files / find_files / move_file / delete_file / get_signed_url** for direct R2 ops on text content. \`grep_file\` runs regex server-side and returns only matching lines + context — reach for it whenever the file might exceed 64KB or you need to find by predicate across many files. \`read_file\` returns a line-number gutter; if you get \`truncated:true\` on a big file, switch to \`grep_file\` or call \`read_file\` again with \`{offset, limit}\` for a deeper slice. \`edit_file\` is the surgical one: read-then-replace a unique anchor, no full-file rewrite. Use it for memory updates and any single-section edit. If your work around the FS call is more than three lines of logic, switch to \`env.FS\` inside exec_code instead. **process_attachment** to understand binary content (PDFs, images) without bloating your context with raw bytes. PDFs go through Sonnet 4.6 server-side rendering, full text plus vision. For modifying binary files (writing PDFs, generating derivatives) use exec_code with a real parser. From cc1047d9e67ab6e1b72551726980804d0db70c45 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 12:15:34 +0530 Subject: [PATCH 11/13] style: oxfmt --- apps/kernel/src/tools/fs-tools.ts | 41 ++++++++++++++++++++++--------- 1 file changed, 29 insertions(+), 12 deletions(-) diff --git a/apps/kernel/src/tools/fs-tools.ts b/apps/kernel/src/tools/fs-tools.ts index f9dcf97..0906399 100644 --- a/apps/kernel/src/tools/fs-tools.ts +++ b/apps/kernel/src/tools/fs-tools.ts @@ -19,12 +19,10 @@ function formatWithLineGutter(text: string, startLine: number): string { const lines = text.split("\n"); // Trailing newline produces an empty final element — preserve it as a // blank line so the rendered output round-trips with the file. - return lines - .map((line, i) => `${startLine + i}\t${line}`) - .join("\n"); + return lines.map((line, i) => `${startLine + i}\t${line}`).join("\n"); } -interface GrepMatch { +export interface GrepMatch { file: string; line: number; match: string; @@ -53,7 +51,10 @@ function scanFileForMatches( regex.lastIndex = 0; if (!regex.test(lines[i]!)) continue; const before = lines.slice(Math.max(0, i - contextLines), i); - const after = lines.slice(i + 1, Math.min(lines.length, i + 1 + contextLines)); + const after = lines.slice( + i + 1, + Math.min(lines.length, i + 1 + contextLines) + ); out.push({ file: filePath, line: i + 1, // 1-based to match read_file's gutter @@ -280,7 +281,9 @@ export const buildFSTools = (perTurn: PerTurnContext) => { }; } const next = - text.slice(0, first) + new_string + text.slice(first + old_string.length); + text.slice(0, first) + + new_string + + text.slice(first + old_string.length); if (next.length > WRITE_CAP) { return { ok: false as const, @@ -433,9 +436,7 @@ export const buildFSTools = (perTurn: PerTurnContext) => { .min(0) .max(10) .optional() - .describe( - "Lines of context before/after each match. Default 2." - ), + .describe("Lines of context before/after each match. Default 2."), max_matches: z .number() .int() @@ -483,7 +484,11 @@ export const buildFSTools = (perTurn: PerTurnContext) => { if (path !== undefined) { const meta = await fs.stat(path); if (!meta) { - return { ok: false as const, error: "not_found" as const, path }; + return { + ok: false as const, + error: "not_found" as const, + path + }; } if (!TEXT_CT_RE.test(meta.contentType)) { return { @@ -497,7 +502,13 @@ export const buildFSTools = (perTurn: PerTurnContext) => { const { content } = await fs.readFile(path); const text = new TextDecoder().decode(content); matches.push( - ...scanFileForMatches(path, text, regex, ctxLines, cap - matches.length) + ...scanFileForMatches( + path, + text, + regex, + ctxLines, + cap - matches.length + ) ); } else { // prefix scope — enumerate files, skip binaries, scan each. @@ -508,7 +519,13 @@ export const buildFSTools = (perTurn: PerTurnContext) => { const { content } = await fs.readFile(f.path); const text = new TextDecoder().decode(content); matches.push( - ...scanFileForMatches(f.path, text, regex, ctxLines, cap - matches.length) + ...scanFileForMatches( + f.path, + text, + regex, + ctxLines, + cap - matches.length + ) ); } } From 7e85f7f379a657fad2c7b27f78b0a8bd20cb4672 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 14:33:20 +0530 Subject: [PATCH 12/13] fix(fs): recognize .ndjson / .jsonl as text MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Smoke caught read_file refusing seed-sample.ndjson as binary. Root cause: .ndjson wasn't in the EXT_CONTENT_TYPE map, so the path-extension sniff fell back to application/octet-stream — and per commit-driver.ts:637 the materialize step always re-sniffs from path on canonical writes, so custom contentTypes passed at writeFile() time don't survive the git-pipeline commit anyway. The only viable fix is the extension map. - env-fs.ts: map ndjson + jsonl → application/x-ndjson. - fs-tools.ts: extend TEXT_CT_RE to accept application/x-ndjson and application/x-jsonl so read_file / grep_file recognize them as text. Co-Authored-By: Claude Opus 4.7 (1M context) --- apps/kernel/src/fs/env-fs.ts | 2 ++ apps/kernel/src/tools/fs-tools.ts | 2 +- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/apps/kernel/src/fs/env-fs.ts b/apps/kernel/src/fs/env-fs.ts index 201094e..0e826a5 100644 --- a/apps/kernel/src/fs/env-fs.ts +++ b/apps/kernel/src/fs/env-fs.ts @@ -196,6 +196,8 @@ const EXT_CONTENT_TYPE: Record = { csv: "text/csv; charset=utf-8", tsv: "text/tab-separated-values; charset=utf-8", json: "application/json", + ndjson: "application/x-ndjson", + jsonl: "application/x-ndjson", yaml: "application/yaml", yml: "application/yaml", xml: "application/xml", diff --git a/apps/kernel/src/tools/fs-tools.ts b/apps/kernel/src/tools/fs-tools.ts index 0906399..338949a 100644 --- a/apps/kernel/src/tools/fs-tools.ts +++ b/apps/kernel/src/tools/fs-tools.ts @@ -6,7 +6,7 @@ import { R2PathError } from "../fs/path"; import { wrappedTool, type PerTurnContext } from "./wrapped-tool"; const TEXT_CT_RE = - /^(text\/|application\/(json|xml|yaml|javascript|typescript|sql|toml|.*\+json|.*\+xml)|image\/svg\+xml)/i; + /^(text\/|application\/(json|xml|yaml|javascript|typescript|sql|toml|x-ndjson|x-jsonl|.*\+json|.*\+xml)|image\/svg\+xml)/i; const READ_CAP = 64 * 1024; const WRITE_CAP = 1024 * 1024; From 7ab0e1ba97c6db6af43a7de81df40d4e8809f7c4 Mon Sep 17 00:00:00 2001 From: Ronit Date: Mon, 18 May 2026 14:43:05 +0530 Subject: [PATCH 13/13] fix(tools): fall back to path-sniff when stored contentType is stale MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Smoke caught a second issue: even after fixing the EXT_CONTENT_TYPE map, canonical R2 holds the old "application/octet-stream" httpMetadata for files written before the fix. materializeMainToCanonical only writes canonical when the git blob sha changes — same-content rewrites don't refresh stored metadata. So a fresh sniff returns x-ndjson, but the stored type stays binary, and the tool's gate rejects. Adds `resolveTextContentType(stored, path)`: trust stored if it's already text-shaped; otherwise re-sniff from the path extension and prefer that when it lands in TEXT_CT_RE. Applied to read_file, edit_file, and grep_file (both single-path and prefix-loop branches). The error path still surfaces the stored type so the agent's reasoning matches what the agent sees in stat() output. Co-Authored-By: Claude Opus 4.7 (1M context) --- apps/kernel/src/tools/fs-tools.ts | 41 +++++++++++++++++++++++++------ 1 file changed, 34 insertions(+), 7 deletions(-) diff --git a/apps/kernel/src/tools/fs-tools.ts b/apps/kernel/src/tools/fs-tools.ts index 338949a..067ee49 100644 --- a/apps/kernel/src/tools/fs-tools.ts +++ b/apps/kernel/src/tools/fs-tools.ts @@ -1,7 +1,11 @@ import { z } from "zod"; import { buildKernelEnvFS } from "../fs/build-kernel-env-fs"; -import { EnvFSError, type FileMeta } from "../fs/env-fs"; +import { + EnvFSError, + sniffContentTypeFromKey, + type FileMeta +} from "../fs/env-fs"; import { R2PathError } from "../fs/path"; import { wrappedTool, type PerTurnContext } from "./wrapped-tool"; @@ -10,6 +14,24 @@ const TEXT_CT_RE = const READ_CAP = 64 * 1024; const WRITE_CAP = 1024 * 1024; +/** + * Pick the contentType we'll trust for the text-or-binary gate. If R2's + * stored httpMetadata is already text-shaped, keep it. Otherwise re-sniff + * from the path extension and prefer that if it lands in TEXT_CT_RE. + * + * Why: `materializeMainToCanonical` in commit-driver.ts re-sniffs from path + * on every write, but skips writes when the git blob sha is unchanged. A + * file written before its extension was added to EXT_CONTENT_TYPE keeps a + * stale `application/octet-stream` on canonical R2 even after the sniff is + * fixed; only a content-change rewrite would refresh it. Path-based fallback + * makes the tool resilient to that drift. + */ +function resolveTextContentType(storedCT: string, path: string): string { + if (TEXT_CT_RE.test(storedCT)) return storedCT; + const sniffed = sniffContentTypeFromKey(path); + return TEXT_CT_RE.test(sniffed) ? sniffed : storedCT; +} + /** * Render text as `\t` per line so the agent can address * sub-ranges by number in follow-up read_file / grep_file calls. @@ -107,7 +129,8 @@ export const buildFSTools = (perTurn: PerTurnContext) => { if (!meta) { return { ok: false as const, error: "not_found" as const, path }; } - if (!TEXT_CT_RE.test(meta.contentType)) { + const effectiveCT = resolveTextContentType(meta.contentType, path); + if (!TEXT_CT_RE.test(effectiveCT)) { return { ok: false as const, error: "binary_not_supported" as const, @@ -117,7 +140,8 @@ export const buildFSTools = (perTurn: PerTurnContext) => { size: meta.size }; } - const { content, contentType, size } = await fs.readFile(path); + const { content, size } = await fs.readFile(path); + const contentType = effectiveCT; const text = new TextDecoder().decode(content); // Explicit-slice path: offset/limit set. Bypass the 64KB cap; the @@ -249,7 +273,8 @@ export const buildFSTools = (perTurn: PerTurnContext) => { if (!meta) { return { ok: false as const, error: "not_found" as const, path }; } - if (!TEXT_CT_RE.test(meta.contentType)) { + const effectiveCT = resolveTextContentType(meta.contentType, path); + if (!TEXT_CT_RE.test(effectiveCT)) { return { ok: false as const, error: "binary_not_supported" as const, @@ -258,7 +283,8 @@ export const buildFSTools = (perTurn: PerTurnContext) => { contentType: meta.contentType }; } - const { content, contentType } = await fs.readFile(path); + const { content } = await fs.readFile(path); + const contentType = effectiveCT; const text = new TextDecoder().decode(content); const first = text.indexOf(old_string); if (first === -1) { @@ -490,7 +516,8 @@ export const buildFSTools = (perTurn: PerTurnContext) => { path }; } - if (!TEXT_CT_RE.test(meta.contentType)) { + const effectiveCT = resolveTextContentType(meta.contentType, path); + if (!TEXT_CT_RE.test(effectiveCT)) { return { ok: false as const, error: "binary_not_supported" as const, @@ -515,7 +542,7 @@ export const buildFSTools = (perTurn: PerTurnContext) => { const files = await fs.listFiles(prefix!); for (const f of files) { if (matches.length >= cap) break; - if (!TEXT_CT_RE.test(f.contentType)) continue; + if (!TEXT_CT_RE.test(resolveTextContentType(f.contentType, f.path))) continue; const { content } = await fs.readFile(f.path); const text = new TextDecoder().decode(content); matches.push(