diff --git a/CLAUDE.md b/CLAUDE.md index 8c59b47..747b779 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -43,7 +43,14 @@ behind every decision. - `src/tools/` — `define`, `prompted` (catalog + dispatch), `mode`, `types`. - `src/agent/` — `schemas`, `planner`/`executor`/`replanner`/`synthesizer`, `runner` (createAgent), `loop-types`. -- `src/memory/` — `store`, `sessions` (IndexedDBStore), `compress`. +- `src/memory/` — `store`, `sessions` (IndexedDBStore), `compress` (compaction). +- `src/tools/approval.ts` (consent gate), `search.ts` (find_tools / large + catalogues), `wrap.ts` (per-run tool wrapping: gate, call budget, output cap). +- `src/subagent/` — `tool` (createSubagentTool), `worker` (serveSubagentWorker), + `protocol` (postMessage shapes). +- `src/{thinking,limits,caching,skills,images}.ts` — see docs/capabilities.md. +- `src/files/` — `vfs` (VirtualFileSystem, IndexedDB `files` store) + `tools` + (createFileTools). - `src/mcp/` — optional HTTP connector (`./mcp` subpath): `http` (transport, mount, refresh) + `oauth` (OAuth 2.1 / DCR provider, vault-backed tokens). - `src/{parse,prompts,events,config,index}.ts`. diff --git a/README.md b/README.md index 57f64f1..2b33d9d 100644 --- a/README.md +++ b/README.md @@ -16,6 +16,16 @@ agent — with host-defined **tools**, optional **MCP**, **system prompts**, **IndexedDB** storage, and an **encrypted token vault**. UI-agnostic: it streams typed events; you render them however you like. +The agent loop also has **thinking** (portable reasoning levels or exact +budgets), **token limits** (input / output / thinking / total per run), +**context compaction** (automatic and `agent.compact()`), **prompt caching**, +**skills** (SKILL.md bundles), **tool consent modes with an autopilot switch**, +**tool search** for catalogues of hundreds of MCP tools, **subagents** in +Web Workers, **image / PDF / file / URL input** (with a clear error for models +that can't take them), +and a **virtual file system** in IndexedDB with `fs_*` tools — see +[docs/capabilities.md](docs/capabilities.md). + [![npm](https://img.shields.io/npm/v/@dudko.dev/agent-web.svg)](https://www.npmjs.com/package/@dudko.dev/agent-web) [![npm](https://img.shields.io/npm/dy/@dudko.dev/agent-web.svg)](https://www.npmjs.com/package/@dudko.dev/agent-web) [![NpmLicense](https://img.shields.io/npm/l/@dudko.dev/agent-web.svg)](https://www.npmjs.com/package/@dudko.dev/agent-web) @@ -191,6 +201,13 @@ Browsers can only speak the **HTTP (StreamableHTTP)** transport — stdio MCP is Node-only. The connector lives in the `./mcp` subpath so the MCP SDK never enters your core bundle. +Pass several servers at once — they connect concurrently, each with its own +deadline (`connectTimeoutMs`, default 30 s), every page of a paginated +`tools/list` is read, and tools a server marks `readOnlyHint` are treated as +read-only by the consent gate. A failing server is reported in `results`; the +others still mount. With hundreds of tools the agent switches to tool search +automatically (see [docs/capabilities.md](docs/capabilities.md#large-tool-catalogues-several-mcp-servers)). + `connectMcpHttp` returns `{ tools, catalog, results, refreshServer, close }`. `tools` and `catalog` are mutated **in place** by `refreshServer(name)`, so an agent built from them picks up a server's new tool list (react to it via the @@ -270,6 +287,35 @@ browser's silence: it re-probes the endpoint with a plain request and, if that gets through, says which header is being refused. `diagnoseMcpCors(url)` is exported so a UI can show the same sentence. +## Thinking, limits, consent, skills, subagents + +```ts +import { createAgent, createSubagentTool, defineSkill } from '@dudko.dev/agent-web' + +const agent = await createAgent({ + model, + tools: { ...tools, ...mcp.tools }, + thinking: 'high', // or { level, budgetTokens }; per stage via stageThinking + limits: { maxTotalTokens: 200_000, maxReasoningTokens: 20_000 }, + maxToolCalls: 40, + compaction: { contextWindowTokens: 128_000 }, // auto-compacts history and long runs + skills: [defineSkill({ name: 'triage', description: 'Triage a bug report', content: '…' })], + toolApproval: { + mode: 'ask-writes', // autopilot | ask-writes | ask-all | read-only + onRequest: async (req) => window.confirm(`Allow ${req.toolName}?`), + }, +}) +agent.setToolApprovalMode('autopilot') // flip the switch at any time +await agent.compact() // summarise the stored transcript now +``` + +With more than 40 tools (several MCP servers) the executor switches to **tool +search**: it starts each step small and calls `find_tools` to activate what it +needs. Subagents are tools too — `createSubagentTool({ config })` runs a child +agent in-process, `createSubagentTool({ worker, workerConfig })` isolates it in a +Web Worker served by `serveSubagentWorker()`. Details, defaults and events: +**[docs/capabilities.md](docs/capabilities.md)**. + ## Configuration (highlights) | Option | Default | Purpose | @@ -280,6 +326,15 @@ exported so a UI can show the same sentence. | `tools` | `{}` | host tools (`defineTool`) | | `availableTools` / `excludedTools` | — | whitelist / blacklist of tool names mounted from `tools` | | `toolMode` | `'auto'` | `native` \| `prompted` \| `auto` (cloud→native, local→prompted) | +| `toolSelectionStrategy` | `'auto'` | `all` \| `plan-narrowed` \| `search` \| `auto` (search above `toolSearchThreshold`, 40) | +| `toolApproval` | autopilot | consent policy: `{ mode, rules, onRequest, timeoutMs }`; `agent.setToolApprovalMode()` | +| `skills` | — | SKILL.md bundles (`defineSkill` / `parseSkillMarkdown` / `loadSkillFromUrl`) | +| `thinking` / `stageThinking` | provider default | `true`, a level (`'low'`…`'xhigh'`, `'none'`), or `{ level, budgetTokens, includeThoughts }` | +| `limits` | — | run caps `maxInputTokens` / `maxOutputTokens` / `maxReasoningTokens` / `maxTotalTokens` + `perCall` output caps | +| `maxToolCalls` / `maxPlanSteps` | ∞ / 8 | tool calls per run / steps per plan | +| `compaction` | auto, ½ of 128k | `{ auto, contextWindowTokens, thresholdTokens, keepRecentTurns, keepRecentSteps, maxToolOutputChars }` | +| `promptCaching` | `true` | stable system prefixes, Anthropic breakpoints, OpenAI cache key | +| `vision` / `inputs` | inferred | what the model takes as attachments (`run(goal, { images, files })`); see `agent.capabilities` | | `systemPrompt` | — | prepended to every phase | | `describeState` | — | serialize world state into prompt context | | `memory` | — | `ContextStore` (`IndexedDBStore` / `MemoryStore`); recent turns are read back into the planner prompt | @@ -287,11 +342,14 @@ exported so a UI can show the same sentence. | `chatTimeoutMs` | 120000 | per-call watchdog | | `replan` / `synthesize` | `true` | toggle phases | | `replanAfter` | `'failure'` | replan trigger: `'failure'` \| `'always'` \| `(stepResult) => boolean \| Promise` | -| `compressAfterChars` | 12000 | summarize old history past this size | +| `compressAfterChars` | 12000 | legacy: summarize old history past this size (when `compaction` is unset) | ## Docs - [docs/design.md](docs/design.md) — architecture & rationale. +- [docs/capabilities.md](docs/capabilities.md) — thinking, limits, compaction, + caching, skills, tool consent / autopilot, large MCP catalogues, subagents, + autonomy, event reference. - [docs/providers.md](docs/providers.md) — every provider, CORS & direct-vs-proxy. - [docs/security.md](docs/security.md) — the token vault & its threat model. - [docs/tasks.md](docs/tasks.md) — status & roadmap. diff --git a/docs/capabilities.md b/docs/capabilities.md new file mode 100644 index 0000000..7041367 --- /dev/null +++ b/docs/capabilities.md @@ -0,0 +1,437 @@ +# Agent capabilities + +Everything the agent loop does beyond "plan → execute → replan → synthesize": +thinking, token budgets, context compaction, skills, tool consent and +autopilot, large MCP catalogues, prompt caching, subagents and autonomy. The +Node sibling [`@dudko.dev/agent`](https://www.npmjs.com/package/@dudko.dev/agent) +implements the same features with the same field and event names. + +- [Thinking](#thinking) +- [Token limits and step caps](#token-limits-and-step-caps) +- [Context management and compaction](#context-management-and-compaction) +- [Prompt caching](#prompt-caching) +- [Token efficiency](#token-efficiency) +- [Skills](#skills) +- [Tool consent and autopilot](#tool-consent-and-autopilot) +- [Large tool catalogues (several MCP servers)](#large-tool-catalogues-several-mcp-servers) +- [Subagents (in-process or Web Workers)](#subagents-in-process-or-web-workers) +- [Images, PDFs, files and URLs in](#images-pdfs-files-and-urls-in--and-models-that-cant-take-them) +- [Virtual file system](#virtual-file-system) +- [Autonomy](#autonomy) +- [Event reference](#event-reference) + +## Thinking + +```ts +createAgent({ + model, + thinking: 'high', // true | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'none' + stageThinking: { executor: 'low', synthesizer: false }, // per stage; wins over `thinking` +}) +// exact budget for providers that take one (Anthropic, Gemini): +createAgent({ model, thinking: { level: 'high', budgetTokens: 8_000, includeThoughts: true } }) +``` + +- The level is the AI SDK's portable `reasoning` setting; each provider maps it + to its native API (OpenAI reasoning effort, Anthropic adaptive thinking, + Gemini thinking level, xAI, DeepSeek, …). Providers without reasoning ignore it + with a warning. +- `budgetTokens` becomes `anthropic.thinking.budgetTokens` and + `google.thinkingConfig.thinkingBudget` (these win over the level for those + providers, by the SDK's precedence rules). +- `includeThoughts` (default `true`) asks Gemini for thought summaries and OpenAI + for a reasoning summary, so thoughts can be streamed. +- `false` / unset sends nothing — the provider default applies. +- Thoughts stream as `step.reasoning-delta` (executor) and + `final.reasoning-delta` (synthesizer); thinking tokens are reported in + `usage.reasoningTokens`. +- Local reasoning models (Qwen3, R1 distills) that inline `…` in + their text have those blocks stripped from parsed output and the answer. + +`resolveThinking(setting)` is exported if you call the AI SDK yourself. + +## Token limits and step caps + +```ts +createAgent({ + model, + limits: { + maxInputTokens: 200_000, // cumulative per run + maxOutputTokens: 20_000, // cumulative per run, thinking included + maxReasoningTokens: 10_000, + maxTotalTokens: 220_000, + perCall: { planner: 800, executor: 1200, replanner: 400, synthesizer: 600, compaction: 1024 }, + }, + maxIterations: 8, // executed steps per run (incl. after a revise) + maxStepsPerTask: 4, // tool-calling rounds inside one step + maxRevisions: 2, // replanner "revise" decisions per run + maxToolCalls: 30, // tool calls per run + maxPlanSteps: 8, // steps in one plan +}) +``` + +Run-level caps are **soft**: they are checked between steps and at every +tool-calling round inside a step, so a run stops at the next boundary after +crossing one, emits `budget.exceeded { kind, tokens, cap }`, and still writes its +answer (bounded by `perCall.synthesizer`). `RunResult.budgetExceeded` tells you +which cap ended the run. `checkLimits(usage, limits)` is exported. + +`usage` carries `inputTokens`, `outputTokens`, `totalTokens`, plus +`reasoningTokens`, `cachedInputTokens` and `cacheWriteTokens` when the provider +reports them. The legacy `budgets` option still works; `limits.perCall` wins. + +## Context management and compaction + +What each stage sees: + +| Stage | System prompt (stable, cached) | User prompt (dynamic) | +| --- | --- | --- | +| Planner | role, skills index, tool catalogue | goal, recent conversation, `describeState` | +| Executor | role, skills index, active skill instructions | goal, state, earlier steps **with their results**, current step (+ catalogue in prompted mode) | +| Replanner | role, active skills | goal, state, progress, remaining plan | +| Synthesizer | role, active skills | goal, state, progress, **tool findings** | + +Each executed step contributes its executor summary (the data it found) to every +later step and to the answer, and the synthesizer also gets clipped excerpts of +successful tool results — so a "find X, then do Y with X" run, or a question +answered from an MCP server, works without `describeState`. + +Compaction keeps long sessions and long runs inside the window: + +```ts +createAgent({ + model, + memory: new IndexedDBStore(), + compaction: { + auto: true, // default + contextWindowTokens: 128_000, + thresholdTokens: 64_000, // default: half the window + keepRecentTurns: 4, // transcript messages kept verbatim + keepRecentSteps: 3, // run steps kept verbatim + summaryMaxTokens: 1024, + maxToolOutputChars: 20_000, // per tool result, as the MODEL sees it + clearToolResultsAfterTokens: 32_000, // default: a quarter of the window; 0 = never + keepToolResults: 3, // newest tool results kept verbatim when clearing + }, +}) +await agent.compact() // manual: summarise the stored transcript now +``` + +- **History** — before planning (and after the run) a transcript above the + threshold is summarised into one message, keeping the last `keepRecentTurns`. +- **Run trace** — between steps a step log above the threshold is folded into a + summary, keeping the last `keepRecentSteps`. +- **Tool results** — every tool's model-facing output is capped at + `maxToolOutputChars` (via the AI SDK `toModelOutput`); events and the trace + still get the full result. +- **Stale tool results inside a step** (context editing, like Anthropic's + `clear_tool_uses` and Claude Code's micro-compaction) — once a step's tool + loop grows past `clearToolResultsAfterTokens`, the oldest results are replaced + with a one-line stub ("… result cleared — call the tool again if you still + need it"), keeping the newest `keepToolResults`. The calls stay, so the model + knows what it did. Clearing is sticky, so the cached prefix doesn't flip back. + +Each compaction emits `context.compacted { scope, beforeTokens, afterTokens }` +(`scope`: `'history' | 'trace' | 'tool-results'`) and, when a model summarised, +a `usage` event with phase `'compact'`. Without `compaction`, the legacy +`compressAfterChars` post-run compression applies as before. + +## Prompt caching + +`promptCaching` is on by default: + +- every stage's **system** prompt holds only run-stable content, the dynamic + parts go to the user prompt — OpenAI's and Gemini's automatic prefix caches hit; +- the system message carries an Anthropic `cacheControl` breakpoint + (`promptCaching: { ttl: '1h' }` for the longer TTL); +- OpenAI calls get a `promptCacheKey` (`:`, or your `key`); +- inside a step's tool loop, a second breakpoint **rolls to the newest message** + every round, so each round reads all earlier rounds from the cache and writes + only the new tail (older message breakpoints are removed — a request carries + at most two, Anthropic allows four); +- tools keep **declaration order** (hosts merge their tool sets in a fixed + order, `useMcpServers` in list order), so the tool list — the head of every + cached prefix — is identical from run to run, without reordering what a server + declared: small models were measured to depend on that order. + +Cache reads/writes show up as `usage.cachedInputTokens` / `cacheWriteTokens`. +`promptCaching: false` sends plain system strings. + +## Token efficiency + +The agent follows the practices coding agents (Claude Code, the Claude and +GitHub Copilot extensions) use to keep a long session cheap: + +| Practice | How the agent does it | Knob | +| --- | --- | --- | +| Stable, cacheable prefix | system prompts hold only run-stable content; deterministic tool order; dynamic state goes last | `promptCaching` | +| Cache the growing loop | rolling Anthropic breakpoint on the newest message; OpenAI `promptCacheKey` per stage | `promptCaching: { ttl }` | +| Don't send every tool | above `toolSearchThreshold` the model gets a compact catalogue and `find_tools` (deferred tools) | `toolSelectionStrategy`, `toolSearchThreshold` | +| Load instructions on demand | skills: only name + description in the prompt, the body on activation | `skills` | +| Cap tool output | per result, as the model sees it (events keep the full value) | `compaction.maxToolOutputChars` | +| Clear stale tool results | oldest results in a long loop become stubs | `compaction.clearToolResultsAfterTokens` | +| Auto-compact | history and run trace summarised past a threshold; `agent.compact()` on demand | `compaction` | +| Isolate side quests | subagents run in their own context; only their answer returns | `createSubagentTool` | +| Pass results, not transcripts | later steps and the answer get step summaries + clipped findings, not raw loops | — | +| Bound everything | token caps per run / per kind, tool-call and plan-step caps | `limits`, `maxToolCalls`, `maxPlanSteps` | +| Think only where it pays | thinking per stage (e.g. on for the planner, off for the synthesizer) | `stageThinking` | +| Cheaper model for summaries | compaction and the answer use the `synthesizer` stage model | `synthesizer` | +| Small images | the React composer downsizes images before sending (≈1.15 MP), the size providers bill for | `` | + +The usage events break every call down by kind (input / output / thinking / +cache read / cache write) and by phase, so the effect is measurable: +`cachedInputTokens / inputTokens` is the cache hit rate. + +## Skills + +Skills are reusable instruction bundles in the [agentskills.io](https://agentskills.io/) +shape (a `SKILL.md` with `name` / `description` frontmatter, a markdown body, +optional bundled files). + +```ts +import { createAgent, defineSkill, parseSkillMarkdown, loadSkillFromUrl } from '@dudko.dev/agent-web' + +const skills = [ + defineSkill({ name: 'release-notes', description: 'Write release notes', content: '…' }), + parseSkillMarkdown(markdownText, [{ path: 'template.md', content: '…' }]), + await loadSkillFromUrl('/skills/triage/SKILL.md', { files: ['labels.md'] }), +] +createAgent({ model, skills }) +``` + +Progressive disclosure: + +1. Only the index (`- name: description`) sits in the planner/executor system prompts. +2. The planner lists the skills that apply (`plan.skills`); their full + instructions enter the executor, replanner and synthesizer prompts. +3. The executor can pull one in later with the built-in `load_skill` tool and + read bundled files with `read_skill_file` (both read-only, never gated). + +`skill.activated { name, by: 'plan' | 'tool' }` is emitted; `RunResult.skills` +lists the skills used; `agent.skills` lists the configured ones. + +## Tool consent and autopilot + +```ts +const agent = await createAgent({ + model, + tools, + toolApproval: { + mode: 'ask-writes', // 'autopilot' (default) | 'ask-writes' | 'ask-all' | 'read-only' + rules: { 'github__delete_*': 'deny', 'docs__*': 'allow' }, // exact names or * globs + onRequest: async (req) => ({ approved: await confirm(`Run ${req.toolName}?`), remember: false }), + timeoutMs: 120_000, // no answer → deny + }, +}) +agent.setToolApprovalMode('autopilot') // the "autopilot" switch, live — even mid-run +``` + +| Mode | Read-only tools | Other tools | +| --- | --- | --- | +| `autopilot` | run | run | +| `ask-writes` | run | ask | +| `ask-all` | ask | ask | +| `read-only` | run | refused (no prompt) | + +Decision order per call: a tool the user chose to **always allow** +(`remember: true`) → the most specific `rules` entry (an exact name, else the +longest glob — rules apply even under autopilot) → the mode. A tool is +read-only when its MCP server says so (`annotations.readOnlyHint`) or you marked +it: `defineTool({ readOnly: true, … })` / `markReadOnly(tool)`. Unknown counts as +"may write". + +A denied call fails with `ToolDeniedError` ("…do not retry it…"), so the model +sees the refusal and the replanner can route around it. Events: +`tool.approval-requested { id, name, input, readOnly }` (only when someone is +asked) and `tool.approval-resolved { id, name, approved, reason, automatic }`. +The gate runs inside each tool's `execute`, so it covers native and prompted +tool-calling, MCP tools and host tools called by subagents alike. + +## Large tool catalogues (several MCP servers) + +Several MCP servers easily add up to hundreds of tools. Sending every schema on +every call wastes the window — and OpenAI rejects more than 128 tools. + +**Loading** (`@dudko.dev/agent-web/mcp`): + +- servers connect **concurrently**; one failing server is reported in `results` + and the others still mount; +- `tools/list` is **paginated** — every `nextCursor` page is followed; +- each server has a **deadline** (`connectTimeoutMs`, default 30 s, per server + or per `connectMcpHttp` call) — a hanging server can't block the rest; +- names are `server__tool`, sanitised to `[a-zA-Z0-9_-]` and capped at 64 + chars; collisions get a `_2` suffix; +- `annotations.readOnlyHint` marks tools read-only for the consent gate; +- `notifications/tools/list_changed` → `onToolsChanged(server)` → `refreshServer`. + +**Selection** — `toolSelectionStrategy` (default `'auto'`): + +| Strategy | Executor sees | Planner sees | +| --- | --- | --- | +| `all` | every tool | full catalogue | +| `plan-narrowed` | the step's `suggestedTools` | full catalogue | +| `search` | built-ins + the step's suggested tools + tools discovered earlier, plus `find_tools` | condensed catalogue grouped by server | +| `auto` | `all` up to `toolSearchThreshold` (40) tools, `search` above | — | + +`find_tools({ query, server?, limit? })` ranks the catalogue by keywords (name +> server > description, prefix matching, no model call) and activates the hits +for the rest of the run (`tools.discovered` event). In prompted mode the newly +available tools — with parameter hints derived from their JSON schemas — are +listed in the next round's prompt. `searchTools(catalog, query)` is exported. + +## Subagents (in-process or Web Workers) + +A subagent is a tool: the parent delegates `{ task }`, the child runs its own +loop and its answer is the tool result. Several delegations in one model step +run in parallel (`maxConcurrent`, default 4). + +```ts +import { createSubagentTool } from '@dudko.dev/agent-web' + +// In-process (shares the page's thread and models, e.g. a loaded WebLLM engine): +const researcher = createSubagentTool({ + name: 'researcher', + description: 'Research a question with the docs tools and report the facts.', + config: { model, tools: docsTools, maxIterations: 4 }, +}) + +// Isolated in a Web Worker: +const analyst = createSubagentTool({ + name: 'analyst', + description: 'Analyse a chess position deeply.', + worker: () => new Worker(new URL('./analyst.worker.ts', import.meta.url), { type: 'module' }), + workerConfig: { model: { providerType: 'google', model: 'gemini-3.5-flash', credentialRef: 'google' } }, + credentials, // resolves the key in the main thread, per task + tools: { get_board }, // host tools, called back over RPC + timeoutMs: 60_000, +}) +createAgent({ model, tools: { researcher, analyst } }) +``` + +```ts +// analyst.worker.ts +import { serveSubagentWorker, defineTool } from '@dudko.dev/agent-web' +import { createGoogleGenerativeAI } from '@ai-sdk/google' + +serveSubagentWorker({ + // Bundlers can't resolve the core's dynamic provider imports in a worker — + // import the factory statically and build the model here. + resolveModel: (spec) => createGoogleGenerativeAI({ apiKey: spec.apiKey })(spec.model), + tools: { deep_search: defineTool({ /* CPU-heavy, runs off the main thread */ }) }, +}) +``` + +- **Worker-local tools** run in the worker — heavy computation never blocks the UI. +- **Host tools** (`tools`) stay in the main thread and are proxied over + `postMessage`; every such call passes the **parent's** consent gate. +- The parent's abort signal and `timeoutMs` stop the child; a worker is + terminated after each task. +- Events: `subagent.start`, `subagent.event` (the child's events, verbatim), + `subagent.complete { text, usage }`, `subagent.error`. Child tokens are + charged to the parent (`usage` with phase `'subagent'`), so the parent's limits + apply. +- The protocol is plain structured-clone messages over any `MessageEndpoint` + (a `Worker`, a worker's `self`, a `MessagePort`). + +## Images, PDFs, files and URLs in — and models that can't take them + +```ts +await agent.run('What is wrong on this screenshot, and does the spec agree?', { + images: [{ data: 'data:image/png;base64,…', name: 'screen.png' }], + files: [ + { data: pdfBytes, mediaType: 'application/pdf', name: 'spec.pdf' }, + { data: 'https://example.com/diagram.png' }, // a link: sent as a URL + ], +}) +``` + +Attachments reach the planner, the executor (every step) and the synthesizer as +file parts of the user message (`toFilePart`). Data can be bytes, base64, a +data URL, or an **http(s) URL** — providers that accept links (Gemini, Claude, +OpenAI) fetch it themselves; for the rest the AI SDK downloads it (from a +browser: subject to CORS). The kind comes from the media type +(`attachmentKind`): `image`, `pdf`, or `file`. + +What the model takes is `agent.capabilities` — `{ images, pdf, files }`, each +`true` / `false` / `undefined`: + +| | images | pdf | other files | +| --- | --- | --- | --- | +| Gemini, Claude, OpenAI GPT-4o/4.1/5 | ✓ | ✓ | tried | +| xAI Grok | ✓ | tried | tried | +| DeepSeek, `gpt-3.5`, `o1-mini`… | ✗ | ✗ | tried | +| Local / prompted mode (WebLLM, built-in AI) | ✗ | ✗ | ✗ | +| OpenAI-compatible servers | tried | tried | tried | + +Override with `vision` (images) or `inputs: { images, pdf, files }`. + +- **Known ✗** — the run ends immediately, no tokens spent, with an `error` + event (phase `run`) and `RunResult.final` set to a message the user can act + on: *The model "…" can't take PDF files. Convert the PDF to text first (e.g. + PDF → Markdown), or switch to a model that reads PDFs such as Gemini, Claude + or GPT-4o/GPT-5.* (`AttachmentsNotSupportedError`; `ImagesNotSupportedError` + for images.) +- **Tried** — a provider refusal is turned into the same message ("the provider + said: …") instead of a raw API error. + +A UI should check `agent.capabilities` before accepting a paste or an upload +(`@dudko.dev/agent-web-react`'s composer does, and can convert a PDF to +Markdown first). Memory stores a text note of the attachments ("[attached: +screen.png, spec.pdf]"), never their bytes. + +## Virtual file system + +```ts +import { VirtualFileSystem, createFileTools } from '@dudko.dev/agent-web' + +const vfs = new VirtualFileSystem() // IndexedDB; { memory: true } for tests / private mode +await vfs.write('/notes/todo.md', '- ship it') +await vfs.writeDataUrl('/img/screen.png', dataUrl) +createAgent({ model, tools: { ...tools, ...createFileTools(vfs) } }) // fs_list, fs_read, fs_write, fs_delete +``` + +A browser workspace shared by the user and the agent: attachments land there, +the agent reads them and writes reports back. Paths are absolute and POSIX-like +(`..` cannot escape the root); text is stored as UTF-8, binary as base64 with a +MIME type; `list(prefix)`, `read`, `readDataUrl`, `delete`, `clear`, and +`onChange(listener)` for live UIs. `namespace` isolates several file systems +in one database; `maxFileBytes` (10 MB) caps a file. `fs_list` / `fs_read` are +read-only (no consent prompt in `ask-writes`); `fs_write` / `fs_delete` are +not; `createFileTools(vfs, { readOnly: true })` mounts only the readers. The +files live in the shared IndexedDB database (`files` store, schema v2 — an +existing v1 database is upgraded in place). + +## Autonomy + +The default prompts make the agent act on its own: + +- the planner plans tool use for anything the tools can do **or look up** — + questions that need data are real goals; an empty plan is reserved for pure + greetings / small talk / questions answerable without tools; +- no stage ever asks the user for confirmation or plans an "ask the user" step; + ambiguity is resolved by the most reasonable interpretation, stated as an + assumption; +- the executor uses `[BLOCKER]` only when a step is truly impossible (a tool is + missing or denied, credentials or data are unavailable); +- the replanner prefers routing around a failure over giving up; +- permission is the host's `toolApproval` policy — never a question in text. + +Soften it per app with `systemPrompt`, or replace any phase via `prompts`. + +## Event reference + +New events (in addition to plan/step/replan/final/usage/retry/stopped/error): + +| Event | Payload | +| --- | --- | +| `step.reasoning-delta` | `step, delta` | +| `final.reasoning-delta` | `delta` | +| `budget.exceeded` | `kind: 'input' \| 'output' \| 'reasoning' \| 'total' \| 'tool-calls', tokens, cap` | +| `context.compacted` | `scope: 'history' \| 'trace', beforeTokens, afterTokens` | +| `skill.activated` | `name, by: 'plan' \| 'tool'` | +| `tools.discovered` | `step?, query, names` | +| `tool.approval-requested` | `id, name, input, readOnly, step?` | +| `tool.approval-resolved` | `id, name, approved, reason?, automatic` | +| `subagent.start` / `subagent.event` / `subagent.complete` / `subagent.error` | `id, name` + `task` / `event` / `text, usage` / `error` | + +`usage` events can now carry phase `'compact'` and `'subagent'`. diff --git a/docs/design.md b/docs/design.md index 3f1bdea..387540b 100644 --- a/docs/design.md +++ b/docs/design.md @@ -25,7 +25,7 @@ bundling), so unifying them would help nothing and couple two release cadences. ## The one seam: an AI SDK `LanguageModel` -Everything is built on the [Vercel AI SDK](https://ai-sdk.dev) (`ai` v6). The +Everything is built on the [Vercel AI SDK](https://ai-sdk.dev) (`ai` v7). The whole package accepts an AI SDK `LanguageModel`; providers are just different ways of producing one. You give the agent a model in one of two ways: @@ -41,14 +41,15 @@ and "structured output" are just thin wrappers over `generateText` / `streamText` / `generateObject`, and the planning agent sits on top of the same primitives. -## Why AI SDK v6 (not v7) +## Why AI SDK v7 -The ecosystem's `ai` is at v7, but the local-model provider -`@browser-ai/web-llm` still peers `ai@^6` (it implements the v6/`@ai-sdk/provider@3` -model spec, `specificationVersion: 'v3'`). Since local models are a core -requirement, the whole package is pinned to the **coherent v6 stack** -(`ai@^6`, `@ai-sdk/*` v3/v2, `@browser-ai/web-llm@^2`). When browser-ai ships v7 -support we bump together. +The package tracks the current `ai` major (v7) together with the provider +packages and `@browser-ai/*` v3, which peers `ai@^7`. (It was pinned to v6 while +the local-model provider `@browser-ai/web-llm` still peered `ai@^6`; the whole +stack moved together once it shipped v7 support.) v7 also brings what the agent +builds on: the portable `reasoning` setting, `instructions` system messages that +carry provider options (Anthropic cache breakpoints), `prepareStep` for +per-step active tools (tool search), and `toModelOutput` (tool-output caps). ## Modules @@ -71,13 +72,18 @@ src/ │ ├── define.ts defineTool() → an AI SDK tool (+ optional promptHint) │ ├── prompted.ts renderCatalog() + dispatch() (the salvage path) │ ├── mode.ts selectToolMode() — native vs prompted per model +│ ├── approval.ts the consent gate (autopilot / ask-writes / ask-all / read-only) +│ ├── search.ts searchTools() + find_tools for large catalogues +│ ├── wrap.ts per-run wrapping: gate, call budget, model-facing output cap │ └── types.ts AgentTool / AgentToolSet ├── agent/ the plan→execute→replan→synthesize loop │ ├── schemas.ts zod Plan/Replan schemas (native path) │ ├── planner/executor/replanner/synthesizer.ts │ ├── runner.ts createAgent() — orchestration + events │ └── loop-types.ts IPlan / IStepResult / IUsage -├── memory/ store.ts, sessions.ts (IndexedDBStore), compress.ts +├── memory/ store.ts, sessions.ts (IndexedDBStore), compress.ts (compaction) +├── subagent/ tool.ts (createSubagentTool), worker.ts (serveSubagentWorker), protocol.ts +├── thinking.ts limits.ts caching.ts skills.ts — see docs/capabilities.md ├── mcp/ OPTIONAL HTTP MCP connector (./mcp subpath) │ ├── http.ts StreamableHTTP connect / mount / refresh │ └── oauth.ts OAuth 2.1 + DCR client provider (vault-backed) diff --git a/docs/tasks.md b/docs/tasks.md index 20c62d4..c5b4949 100644 --- a/docs/tasks.md +++ b/docs/tasks.md @@ -79,3 +79,32 @@ Status of the work. See [design.md](./design.md) for the architecture, concerns that live in the sibling package `@dudko.dev/agent`. - Shipping shared/app-owned API keys to the client — use a proxy or the gateway (see [security.md](./security.md)). + +## Done — agent capabilities (see [capabilities.md](./capabilities.md)) + +- [x] **Thinking** — `thinking` / `stageThinking` (portable level or exact + budget), streamed thoughts, reasoning tokens in usage. +- [x] **Token limits** — `limits` (input / output / reasoning / total per run, + per-call output caps), `maxToolCalls`, `maxPlanSteps`; soft caps that stop + at the next boundary and still answer. +- [x] **Context** — step results and tool findings flow into later steps and + the answer; auto + manual compaction (history, run trace); model-facing + tool-output cap. +- [x] **Prompt caching** — stable system prefixes, Anthropic breakpoints, + OpenAI cache keys; cached tokens in usage. +- [x] **Skills** — SKILL.md parsing/loading, planner selection, `load_skill` / + `read_skill_file`. +- [x] **Tool consent** — autopilot / ask-writes / ask-all / read-only, glob + rules, "always allow", live `setToolApprovalMode`. +- [x] **Large MCP catalogues** — pagination, per-server deadlines, readOnly + annotations, `search` strategy with `find_tools`. +- [x] **Subagents** — `createSubagentTool` in-process or in a Web Worker + (`serveSubagentWorker`), proxied host tools under the parent's consent gate. +- [x] **Autonomy** — prompts never ask the user; questions that need data are + planned as tool work. +- [x] **Attachments** — `run(goal, { images, files })`: images, PDFs, files, + http(s) URLs as file parts; `agent.capabilities` per kind; + `AttachmentsNotSupportedError` before any call (known text-only models) + or from a provider refusal. +- [x] **Virtual file system** — `VirtualFileSystem` (IndexedDB / memory) + + `createFileTools` (`fs_list` / `fs_read` / `fs_write` / `fs_delete`). diff --git a/package-lock.json b/package-lock.json index dae5c8e..a8d0d9b 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@dudko.dev/agent-web", - "version": "0.0.19", + "version": "0.0.20", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@dudko.dev/agent-web", - "version": "0.0.19", + "version": "0.0.20", "funding": [ { "type": "individual", @@ -27,26 +27,26 @@ ], "license": "MIT", "dependencies": { - "ai": "^7.0.116", - "idb": "^8.0.0", + "ai": "^7.0.127", + "idb": "^8.0.3", "zod": "^4.6.5" }, "devDependencies": { - "@ai-sdk/anthropic": "^4.0.65", - "@ai-sdk/deepseek": "^3.0.54", - "@ai-sdk/google": "^4.0.82", - "@ai-sdk/openai": "^4.0.77", - "@ai-sdk/openai-compatible": "^3.0.57", - "@ai-sdk/xai": "^5.0.10", - "@browser-ai/core": "^3.0.3", - "@browser-ai/web-llm": "^3.0.3", + "@ai-sdk/anthropic": "^4.0.71", + "@ai-sdk/deepseek": "^3.0.58", + "@ai-sdk/google": "^4.0.87", + "@ai-sdk/openai": "^4.0.83", + "@ai-sdk/openai-compatible": "^3.0.62", + "@ai-sdk/xai": "^5.0.14", + "@browser-ai/core": "^3.0.4", + "@browser-ai/web-llm": "^3.0.4", "@mlc-ai/web-llm": "^0.2.85", - "@modelcontextprotocol/sdk": "^1.30.1", + "@modelcontextprotocol/sdk": "^1.32.0", "@types/node": "^22.9.0", "fake-indexeddb": "^6.2.5", "prettier": "^3.9.9", "tsup": "^8.5.1", - "typescript": "^5.9.3" + "typescript": "^6.0.3" }, "engines": { "node": ">=18" @@ -97,14 +97,14 @@ } }, "node_modules/@ai-sdk/anthropic": { - "version": "4.0.65", - "resolved": "https://registry.npmjs.org/@ai-sdk/anthropic/-/anthropic-4.0.65.tgz", - "integrity": "sha512-3fmxxn5xWGOPU1wGrUCI30YFW5r6h2WrY+sz3rokkVlBHrDr889hPYP1yNoSmOVvtAwCFWlm79P2lRngIPOObg==", + "version": "4.0.71", + "resolved": "https://registry.npmjs.org/@ai-sdk/anthropic/-/anthropic-4.0.71.tgz", + "integrity": "sha512-wiv3jhUH0RrvGzSolJMZfZ2cI9As45LhwYmcAoWvNivXjC7wzJ4W/Udq9/u7EGnGsT8lpmFKEKBrMXVeUO97Dg==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -114,14 +114,14 @@ } }, "node_modules/@ai-sdk/deepseek": { - "version": "3.0.54", - "resolved": "https://registry.npmjs.org/@ai-sdk/deepseek/-/deepseek-3.0.54.tgz", - "integrity": "sha512-1ByKYV/uis+3U66kueMJgF2DjN0DcJVtacZf7s1+BttKQwIRmHhKnQ3aqDRlIzpc2as59eqO5dxGi8Bfii2Pnw==", + "version": "3.0.58", + "resolved": "https://registry.npmjs.org/@ai-sdk/deepseek/-/deepseek-3.0.58.tgz", + "integrity": "sha512-tOafL69mHQGJH3g/mu4k0evkP/cRLSHCWe7L9rcVV4b0S1zI2rpWDZbToZO7D4X/uox9baVmgRQ5AoUTeaDs0A==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -131,13 +131,13 @@ } }, "node_modules/@ai-sdk/gateway": { - "version": "4.0.94", - "resolved": "https://registry.npmjs.org/@ai-sdk/gateway/-/gateway-4.0.94.tgz", - "integrity": "sha512-hcp7GLnMynH6NfMnemgvW9acn7j3WsUe/dHFzX3/fhPit08xhm1aYmDGHMiIT6YwsXVBxkit9TriKb2SGAOg+Q==", + "version": "4.0.103", + "resolved": "https://registry.npmjs.org/@ai-sdk/gateway/-/gateway-4.0.103.tgz", + "integrity": "sha512-nnGTUHzPRQ14jhv9CL4OQrzcRBIFTBMIPKGIi9Z9u/4GtwSRxa/nxUX6np2Iwp5phOzO0a0njj/j9Jolf5AEaA==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53", "@vercel/oidc": "3.2.0" }, "engines": { @@ -148,14 +148,14 @@ } }, "node_modules/@ai-sdk/google": { - "version": "4.0.82", - "resolved": "https://registry.npmjs.org/@ai-sdk/google/-/google-4.0.82.tgz", - "integrity": "sha512-2zPYyBSyIcywLtOEJRxlrtXjBjz4KxEU19Qqg/Dk8I9BrsB08+ghG2jv155dtuDZj8qZMrh6LO8J+Wt5O4xVhg==", + "version": "4.0.87", + "resolved": "https://registry.npmjs.org/@ai-sdk/google/-/google-4.0.87.tgz", + "integrity": "sha512-s0tWc+QeSce7O2ZLm+VWSRQy16dGfqEcMIUxEWh+LFvXX3h0TLH8jxCQ2H1qhiBgTta+6A97eHPhBlRFu1xy+g==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -165,14 +165,14 @@ } }, "node_modules/@ai-sdk/openai": { - "version": "4.0.77", - "resolved": "https://registry.npmjs.org/@ai-sdk/openai/-/openai-4.0.77.tgz", - "integrity": "sha512-KUCeRneOIww8az/mH3dy68csmOTqa7s0F6fiHEkdXtr1u0xUoQTIU4y7k0YrjjBuyybpX2yN94rf9elhPu4zFg==", + "version": "4.0.83", + "resolved": "https://registry.npmjs.org/@ai-sdk/openai/-/openai-4.0.83.tgz", + "integrity": "sha512-NgqZVWoya7hfmtLkWyIjX9THbyYMLfxFqpShdAmvMnkoT+R9AZbdULGan8+4BiBWIZ1aHiAsc90l3SL8QU/0+Q==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -182,14 +182,14 @@ } }, "node_modules/@ai-sdk/openai-compatible": { - "version": "3.0.57", - "resolved": "https://registry.npmjs.org/@ai-sdk/openai-compatible/-/openai-compatible-3.0.57.tgz", - "integrity": "sha512-TEY5Ng/bTJXrYy9wpdxK70gABdYejYedaKbmPCZLhBkPe21rmLzkGxAGkfPYgAfCKPaoAO9THrEhmwUZ4Xhg2Q==", + "version": "3.0.62", + "resolved": "https://registry.npmjs.org/@ai-sdk/openai-compatible/-/openai-compatible-3.0.62.tgz", + "integrity": "sha512-IAz1xv8zot4KLGa8kN7UfAA6H5+ARzAgAKLsoKxHPBzZuOc2E3ohaELg0x/JR9HVTbMbiftb0dPyCb8cT2KBRg==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -199,9 +199,9 @@ } }, "node_modules/@ai-sdk/provider": { - "version": "4.0.18", - "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.18.tgz", - "integrity": "sha512-+GZJIgz1jk86pwEbb3f1BD2bdoSKyWE4Jg4YUc7NMnMozbWemSKYZcw2F4nMaO5qwsL5A8RmioAKW81YktRpIQ==", + "version": "4.0.21", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.21.tgz", + "integrity": "sha512-UpbC9C1oht8dhfKPbVXSLRZS3DI8uk8n5v2uMjBPNIYGsC2kL045ywq7D8Hh9KgnH9rP5W/GxQdYtWScRouqlA==", "license": "Apache-2.0", "dependencies": { "json-schema": "^0.4.0" @@ -211,12 +211,12 @@ } }, "node_modules/@ai-sdk/provider-utils": { - "version": "5.0.49", - "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.49.tgz", - "integrity": "sha512-T+/H8DCvqJoCqLhltVatS7ffA779Iyk/ZZrgNXkrgv51E0n2Cd6eIdCmtwhCGIKqtbTtwCvTmQdAMv5Z8rOGHQ==", + "version": "5.0.53", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.53.tgz", + "integrity": "sha512-VVe6UDd0y0/B4TfauuKKzcppqvhTmSluwwUVT4WG2/8h7gFbyFm7PM4UGM+ZXe3/RI6bv+Vl1cw/7aiBoQqdxg==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", + "@ai-sdk/provider": "4.0.21", "@standard-schema/spec": "^1.1.0", "@workflow/serde": "4.1.0", "eventsource-parser": "^3.0.8", @@ -230,14 +230,14 @@ } }, "node_modules/@ai-sdk/xai": { - "version": "5.0.10", - "resolved": "https://registry.npmjs.org/@ai-sdk/xai/-/xai-5.0.10.tgz", - "integrity": "sha512-fSZtjtoQPa6VR6a30WNGagbfIECWTJA1I5dJCovYOC/JDZqibpBNCkvjnxd2YaWJy0Ozzq52rWCS+D1sY3M/uQ==", + "version": "5.0.14", + "resolved": "https://registry.npmjs.org/@ai-sdk/xai/-/xai-5.0.14.tgz", + "integrity": "sha512-Omgf/lFOf1nHhRGlRQqD/t7DPllZaxen1kIGiMrgYfnhtfdM7YcKMUY76rw9CeFrOldGLfewbfShAm9hWAp5FA==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -247,9 +247,9 @@ } }, "node_modules/@browser-ai/core": { - "version": "3.0.3", - "resolved": "https://registry.npmjs.org/@browser-ai/core/-/core-3.0.3.tgz", - "integrity": "sha512-A43vxn4oevqjztdHmwkh+mcTEutDFT2j/z6DPObq/a8BMOZw0mejhTJrul2iRDwlOn+YQjhgEWidXODNww8eoQ==", + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@browser-ai/core/-/core-3.0.4.tgz", + "integrity": "sha512-9A2a3ddfjeNZeLUTZW5CWeER+wn+JdcQCGRgk4Nv/wLIw5/2jxs+TxStcFMRJ1nB3Uiql/zcQdIyQRQvHxEV5Q==", "dev": true, "license": "Apache-2.0", "dependencies": { @@ -260,9 +260,9 @@ } }, "node_modules/@browser-ai/web-llm": { - "version": "3.0.3", - "resolved": "https://registry.npmjs.org/@browser-ai/web-llm/-/web-llm-3.0.3.tgz", - "integrity": "sha512-u14PZEwZW8wDDxi4b1SII6NdONIqGTsDOWHKRCCVpgNN3bE/XZqAi1Tq+TjWlS5wU1tYq6p7GjRNPcHgTAJJyA==", + "version": "3.0.4", + "resolved": "https://registry.npmjs.org/@browser-ai/web-llm/-/web-llm-3.0.4.tgz", + "integrity": "sha512-716/SO/NqrbYjy7IZiBBNGfYB0hu5ObUSdtMc/8BsiiCz6FTE45fwva8WhCJl3QNh1M2q2qBw1T+u8t8abBq2A==", "dev": true, "license": "Apache-2.0", "peerDependencies": { @@ -782,9 +782,9 @@ } }, "node_modules/@modelcontextprotocol/sdk": { - "version": "1.30.1", - "resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.30.1.tgz", - "integrity": "sha512-H2HxLvC3HDNybePJaLdSrU1hhUK5iQw+WvV1b01myFyI7sdVGe1u/IPTE5D9fGCiJDVtgMV/lmFkQXLmQyIFYA==", + "version": "1.32.0", + "resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.32.0.tgz", + "integrity": "sha512-8BviX/hK4Gd2eL1KdTwm0gfl4d3wdoErW0Jd00h6nomUAtk6IJk5vwQJc6kfWtJTzN5OEV/lgIiJKI8fiDGAlA==", "dev": true, "license": "MIT", "dependencies": { @@ -1238,14 +1238,14 @@ } }, "node_modules/ai": { - "version": "7.0.116", - "resolved": "https://registry.npmjs.org/ai/-/ai-7.0.116.tgz", - "integrity": "sha512-gmkVGPzTNPcJixBiG9zUvP3gmvwko85dHiKrxFKK0zcVJIIu1FP6YDo3jna+ZrC2twMAsUPLr5htCwZjCINH3Q==", + "version": "7.0.127", + "resolved": "https://registry.npmjs.org/ai/-/ai-7.0.127.tgz", + "integrity": "sha512-JNsNPk4ZvRGEJuqdVo2lYFkm7uAuf3p/RwKFWn1Wh/+IK3U1RyFWw2GXoFnhzqNfxEuj7XJB4WpIUhl1sHToWg==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/gateway": "4.0.94", - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/gateway": "4.0.103", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -2975,9 +2975,9 @@ } }, "node_modules/typescript": { - "version": "5.9.3", - "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", - "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", + "version": "6.0.3", + "resolved": "https://registry.npmjs.org/typescript/-/typescript-6.0.3.tgz", + "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", "dev": true, "license": "Apache-2.0", "bin": { @@ -2996,9 +2996,9 @@ "license": "MIT" }, "node_modules/undici": { - "version": "7.29.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.0.tgz", - "integrity": "sha512-IDxfleLmmbSskfWSUATiN1nfn2rDuvnMOqb5CWR92iIfojA0Ud+ulOAAEQ57LPr9rWmsreUyf5lwyao+7GNNVw==", + "version": "7.30.0", + "resolved": "https://registry.npmjs.org/undici/-/undici-7.30.0.tgz", + "integrity": "sha512-dkrQXeHSaoamnItlYbmzG0wFYrM0ZwDxCIg0A7aKjTyyhh9svRzCNFEzV+Vm05/yehjCzjDZ31KXfGEjYSztDQ==", "license": "MIT", "engines": { "node": ">=20.18.1" diff --git a/package.json b/package.json index e7cda25..4fa5bd5 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@dudko.dev/agent-web", - "version": "0.0.19", + "version": "0.0.20", "description": "Headless, configurable in-browser LLM agent on the Vercel AI SDK: multi-provider bring-your-own-key (OpenAI, Anthropic, Google, xAI, DeepSeek, gateway, openai-compatible) AND local WebGPU/WebLLM, native + prompted tool-calling, structured output, an encrypted IndexedDB token vault, and a plan → execute → replan → synthesize loop. UI-agnostic.", "type": "module", "sideEffects": false, @@ -125,8 +125,8 @@ "node": ">=18" }, "dependencies": { - "ai": "^7.0.116", - "idb": "^8.0.0", + "ai": "^7.0.127", + "idb": "^8.0.3", "zod": "^4.6.5" }, "peerDependencies": { @@ -174,20 +174,20 @@ } }, "devDependencies": { - "@ai-sdk/anthropic": "^4.0.65", - "@ai-sdk/deepseek": "^3.0.54", - "@ai-sdk/google": "^4.0.82", - "@ai-sdk/openai": "^4.0.77", - "@ai-sdk/openai-compatible": "^3.0.57", - "@ai-sdk/xai": "^5.0.10", - "@browser-ai/core": "^3.0.3", - "@browser-ai/web-llm": "^3.0.3", + "@ai-sdk/anthropic": "^4.0.71", + "@ai-sdk/deepseek": "^3.0.58", + "@ai-sdk/google": "^4.0.87", + "@ai-sdk/openai": "^4.0.83", + "@ai-sdk/openai-compatible": "^3.0.62", + "@ai-sdk/xai": "^5.0.14", + "@browser-ai/core": "^3.0.4", + "@browser-ai/web-llm": "^3.0.4", "@mlc-ai/web-llm": "^0.2.85", - "@modelcontextprotocol/sdk": "^1.30.1", + "@modelcontextprotocol/sdk": "^1.32.0", "@types/node": "^22.9.0", "fake-indexeddb": "^6.2.5", "prettier": "^3.9.9", "tsup": "^8.5.1", - "typescript": "^5.9.3" + "typescript": "^6.0.3" } } diff --git a/src/agent/executor.ts b/src/agent/executor.ts index 3ad6f31..8229562 100644 --- a/src/agent/executor.ts +++ b/src/agent/executor.ts @@ -1,9 +1,11 @@ import type { ReplanTrigger } from '../config.js' import { runToolLoop } from '../llm/tool-loop.js' import { timeoutSignal } from '../llm/util.js' +import { renderActiveSkills, renderSkillIndex } from '../skills.js' import { renderCatalog } from '../tools/prompted.js' +import { FIND_TOOLS_NAME } from '../tools/search.js' import { BLOCKER, type IPlanStep, type IStepResult, type IUsage } from './loop-types.js' -import { systemFor, type AgentContext } from './internal.js' +import { imageRefusal, promptFor, stageCall, type AgentContext } from './internal.js' /** * Split the executor's reply into a clean summary and a structural `blocked` @@ -16,14 +18,26 @@ export const splitBlocker = (raw: string): { summary: string; blocked: boolean } return { summary: raw.split(BLOCKER).join('').trim(), blocked: true } } +/** + * The tool names the executor may call in this step, or undefined for "all". + * 'plan-narrowed' → the step's suggestedTools; 'search' → built-ins, the + * step's suggestedTools and tools discovered earlier in the run (find_tools + * adds more while the step runs). + */ const activeToolNames = (ctx: AgentContext, step: IPlanStep): string[] | undefined => { - if (ctx.config.toolSelectionStrategy !== 'plan-narrowed') return undefined + const strategy = ctx.strategy ?? ctx.config.toolSelectionStrategy + const builtins = [...(ctx.builtinTools ?? [])].filter((n) => ctx.tools[n]) const suggested = step.suggestedTools ?? [] const known = suggested.filter((n) => ctx.tools[n]) + if (strategy === 'search') { + const discovered = (ctx.discovered?.() ?? []).slice(-16) + return [...new Set([...builtins, ...known, ...discovered])] + } + if (strategy !== 'plan-narrowed') return undefined // A step that named tools but all were unknown still wants tools: fall back to // the full set rather than stalling with zero. if (known.length === 0 && suggested.length > 0) return Object.keys(ctx.tools) - return known + return [...new Set([...known, ...builtins])] } const summaryOf = (text: string, toolCallCount: number): string => { @@ -31,6 +45,13 @@ const summaryOf = (text: string, toolCallCount: number): string => { return toolCallCount > 0 ? `Executed ${toolCallCount} tool call(s).` : 'Step produced no output.' } +/** Catalogue lines ("- name(hint): description") for the given tool names. */ +const catalogFor = (ctx: AgentContext, names: string[]): string => + renderCatalog( + Object.fromEntries(names.filter((n) => ctx.tools[n]).map((n) => [n, ctx.tools[n]])), + ctx.toolHint, + ) + /** Execute one plan step: run the tool loop (native or prompted) and emit events. */ export const executeStep = async ( ctx: AgentContext, @@ -39,16 +60,16 @@ export const executeStep = async ( index: number, total: number, done: string[], -): Promise<{ result: IStepResult; usage: IUsage }> => { +): Promise<{ result: IStepResult; usage: IUsage; stoppedByBudget?: boolean }> => { const state = await ctx.state() - // Under plan-narrowed selection the prompted path must also see the narrowed - // catalogue — the prompt is its only tool surface (native mode gets the SDK's - // activeTools instead and never renders the catalogue). - const active = activeToolNames(ctx, step) - const toolCatalog = - active === undefined - ? ctx.toolCatalog - : renderCatalog(Object.fromEntries(active.map((name) => [name, ctx.tools[name]]))) + ctx.setCurrentStep?.(step) + // Under plan-narrowed / search selection the prompted path must also see the + // narrowed catalogue — the prompt is its only tool surface (native mode gets + // the SDK's activeTools instead and never renders the catalogue). + const initial = activeToolNames(ctx, step) + const searchMode = (ctx.strategy ?? ctx.config.toolSelectionStrategy) === 'search' + const toolCatalog = catalogFor(ctx, initial ?? Object.keys(ctx.tools)) + const active = ctx.activeSkills?.() ?? [] const parts = ctx.prompts.executor({ goal, state, @@ -58,21 +79,48 @@ export const executeStep = async ( toolCatalog, done, mode: ctx.executorMode, + skills: ctx.skills?.length ? renderSkillIndex(ctx.skills) : undefined, + activeSkills: active.length ? renderActiveSkills(active) : undefined, + searchMode: searchMode && Boolean(ctx.tools[FIND_TOOLS_NAME]), }) const loop = await runToolLoop(ctx.executorModel, { mode: ctx.executorMode, - system: systemFor(ctx, parts.system), - prompt: parts.prompt, + ...stageCall(ctx, 'executor', parts.system), + ...promptFor(ctx, parts.prompt), tools: ctx.tools, - activeTools: active, + // In search mode the active set grows while the step runs. + activeTools: searchMode + ? () => [...new Set([...(initial ?? []), ...(ctx.discovered?.() ?? [])])] + : initial, + describeTools: (names) => catalogFor(ctx, names), maxSteps: ctx.config.maxStepsPerTask, + shouldStop: ctx.overBudget, maxOutputTokens: ctx.config.budgets.executor, temperature: ctx.config.temperature, abortSignal: ctx.signal, timeoutMs: ctx.config.chatTimeoutMs, + caching: ctx.caching, + ...(ctx.config.compaction.clearToolResultsAfterTokens > 0 + ? { + clearToolResults: { + triggerTokens: ctx.config.compaction.clearToolResultsAfterTokens, + keep: ctx.config.compaction.keepToolResults, + }, + } + : {}), callbacks: { + onToolResultsCleared: (info) => { + ctx.log.info(`cleared ${info.cleared} stale tool result(s) from the step's context`) + ctx.emit({ + type: 'context.compacted', + scope: 'tool-results', + beforeTokens: info.beforeTokens, + afterTokens: info.afterTokens, + }) + }, onTextDelta: (delta) => ctx.emit({ type: 'step.text-delta', step, delta }), + onReasoningDelta: (delta) => ctx.emit({ type: 'step.reasoning-delta', step, delta }), onToolCall: (name, input) => { ctx.log.info(`tool call: ${name}`, input) ctx.emit({ type: 'step.tool-call', step, name, input }) @@ -82,6 +130,8 @@ export const executeStep = async ( ctx.emit({ type: 'step.tool-result', step, name, output, ok }) }, }, + }).catch((err: unknown) => { + throw imageRefusal(ctx, err, ctx.executorModel) ?? err }) ctx.log.debug('executor text:', loop.text) @@ -94,6 +144,7 @@ export const executeStep = async ( blocked, }, usage: loop.usage, + stoppedByBudget: loop.stoppedByBudget, } } diff --git a/src/agent/internal.ts b/src/agent/internal.ts index bd1b07c..f4b48e3 100644 --- a/src/agent/internal.ts +++ b/src/agent/internal.ts @@ -1,11 +1,34 @@ -import type { LanguageModel } from 'ai' +import type { LanguageModel, ModelMessage, SystemModelMessage } from 'ai' +import { cachedInstructions, cachingProviderOptions, type ResolvedCaching } from '../caching.js' import type { BrowserAgentConfig, ResolvedConfig } from '../config.js' import type { AgentEvent } from '../events.js' import type { StoredMessage } from '../memory/store.js' import type { Prompts, ToolCallMode } from '../prompts.js' import { withSystem } from '../prompts.js' +import type { Skill } from '../skills.js' +import { + mergeProviderOptions, + resolveThinking, + thinkingFor, + type ProviderOptionsMap, + type ThinkingLevel, + type ThinkingStage, +} from '../thinking.js' import type { AgentToolSet } from '../tools/types.js' import type { AgentLogger } from '../logger.js' +import { + AttachmentsNotSupportedError, + attachmentKind, + isAttachmentRefusal, + modelLabel, + toFilePart, + unsupportedAttachment, + type RunFile, +} from '../images.js' +import type { IPlanStep, IUsage } from './loop-types.js' + +/** The effective tool-selection strategy of a run ('auto' already resolved). */ +export type EffectiveToolStrategy = 'all' | 'plan-narrowed' | 'search' /** Everything the phase functions (planner/executor/replanner/synthesizer) share. */ export interface AgentContext { @@ -18,7 +41,9 @@ export interface AgentContext { plannerMode: ToolCallMode /** Tool-mode for the executor (from the executor model). */ executorMode: ToolCallMode + /** The run's tools (host + MCP + built-ins), already wrapped by the consent gate. */ tools: AgentToolSet + /** The catalogue as the planner sees it (condensed in search mode). */ toolCatalog: string prompts: Prompts emit: (event: AgentEvent) => void @@ -29,6 +54,70 @@ export interface AgentContext { state: () => Promise /** Session transcript loaded from memory BEFORE this run's goal was appended. */ history?: StoredMessage[] + + /** Effective tool selection for this run. */ + strategy?: EffectiveToolStrategy + /** Prompt caching, resolved. */ + caching?: ResolvedCaching + /** All configured skills. */ + skills?: Skill[] + /** Skills whose instructions are in play for this run. */ + activeSkills?: () => Skill[] + /** Tools find_tools activated during this run (search mode). */ + discovered?: () => string[] + /** Built-in tool names (find_tools, skill tools) — always callable. */ + builtinTools?: ReadonlySet + /** Parameter hint of a tool, for prompted catalogues. */ + toolHint?: (name: string) => string | undefined + /** Run usage so far (for in-step budget checks). */ + usageSoFar?: () => IUsage + /** True when the run's limits are reached given extra usage spent inside the current call. */ + overBudget?: (extra: IUsage) => boolean + /** Tracks the step being executed (tools read it via their run context). */ + setCurrentStep?: (step: IPlanStep | undefined) => void + /** Images / PDFs / files the user sent with this run's goal. */ + images?: RunFile[] +} + +/** + * The prompt of a stage call: plain text, or — when the run carries images — a + * user message with the text and the images, so the model can see them. + */ +export const promptFor = ( + ctx: AgentContext, + prompt: string, +): { prompt: string } | { messages: ModelMessage[] } => + ctx.images?.length + ? { + messages: [ + { + role: 'user', + content: [{ type: 'text', text: prompt }, ...ctx.images.map(toFilePart)], + }, + ], + } + : { prompt } + +/** + * When a run carries attachments and a model call failed in a way that reads + * like a refusal of them, the clear error to raise instead (else undefined). + */ +export const imageRefusal = ( + ctx: AgentContext, + err: unknown, + model: LanguageModel, +): AttachmentsNotSupportedError | undefined => { + if ( + !ctx.images?.length || + err instanceof AttachmentsNotSupportedError || + !isAttachmentRefusal(err) + ) { + return undefined + } + const kinds = new Set(ctx.images.map(attachmentKind)) + const kind = kinds.size === 1 ? [...kinds][0] : 'file' + const detail = err instanceof Error ? err.message.slice(0, 160) : undefined + return unsupportedAttachment(modelLabel(model), kind, detail && `the provider said: ${detail}`) } /** Prepend the host's systemPrompt to a phase system prompt. */ @@ -36,3 +125,33 @@ export const systemFor = (ctx: AgentContext, base: string): string => withSystem(base, ctx.raw.systemPrompt) export const aborted = (ctx: AgentContext): boolean => ctx.signal?.aborted === true + +export interface StageCallOptions { + system: string | SystemModelMessage + reasoning?: ThinkingLevel + providerOptions?: ProviderOptionsMap +} + +/** + * The per-stage call settings: the host systemPrompt + the phase system prompt + * (as a cacheable message when prompt caching is on), the stage's thinking + * level, and the merged provider options (thinking, caching). + */ +export const stageCall = ( + ctx: AgentContext, + stage: ThinkingStage, + baseSystem: string, +): StageCallOptions => { + const system = systemFor(ctx, baseSystem) + const thinking = resolveThinking(thinkingFor(stage, ctx.raw.thinking, ctx.raw.stageThinking)) + const caching = ctx.caching ?? { enabled: false } + const providerOptions = mergeProviderOptions( + cachingProviderOptions(caching, ctx.config.clientName, stage), + thinking.providerOptions, + ) + return { + system: cachedInstructions(system, caching), + ...(thinking.reasoning ? { reasoning: thinking.reasoning } : {}), + ...(providerOptions ? { providerOptions } : {}), + } +} diff --git a/src/agent/loop-types.ts b/src/agent/loop-types.ts index 371bae8..c479591 100644 --- a/src/agent/loop-types.ts +++ b/src/agent/loop-types.ts @@ -18,6 +18,12 @@ export interface IUsage { inputTokens: number outputTokens: number totalTokens: number + /** Output tokens spent thinking (a subset of outputTokens). */ + reasoningTokens?: number + /** Input tokens served from the provider's prompt cache. */ + cachedInputTokens?: number + /** Input tokens written to the provider's prompt cache. */ + cacheWriteTokens?: number } export interface IToolCall { @@ -42,10 +48,20 @@ export interface IStepResult { */ export const BLOCKER = '[BLOCKER]' -export const emptyUsage = (): IUsage => ({ inputTokens: 0, outputTokens: 0, totalTokens: 0 }) +export const emptyUsage = (): IUsage => ({ + inputTokens: 0, + outputTokens: 0, + totalTokens: 0, + reasoningTokens: 0, + cachedInputTokens: 0, + cacheWriteTokens: 0, +}) export const addUsage = (a: IUsage, b: IUsage): IUsage => ({ inputTokens: a.inputTokens + b.inputTokens, outputTokens: a.outputTokens + b.outputTokens, totalTokens: a.totalTokens + b.totalTokens, + reasoningTokens: (a.reasoningTokens ?? 0) + (b.reasoningTokens ?? 0), + cachedInputTokens: (a.cachedInputTokens ?? 0) + (b.cachedInputTokens ?? 0), + cacheWriteTokens: (a.cacheWriteTokens ?? 0) + (b.cacheWriteTokens ?? 0), }) diff --git a/src/agent/planner.ts b/src/agent/planner.ts index 8935979..b750e90 100644 --- a/src/agent/planner.ts +++ b/src/agent/planner.ts @@ -2,15 +2,18 @@ import { generate, generateStructured } from '../llm/generate.js' import { normalizeUsage } from '../llm/util.js' import { parsePlannerResponse } from '../parse.js' import type { ToolCallMode } from '../prompts.js' +import { renderSkillIndex } from '../skills.js' import type { IPlan, IUsage } from './loop-types.js' -import { systemFor, type AgentContext } from './internal.js' +import { imageRefusal, promptFor, stageCall, type AgentContext } from './internal.js' import { PlanSchema } from './schemas.js' const toSteps = ( raw: { description: string; expectedOutcome?: string; suggestedTools?: string[] }[], + cap: number, ): IPlan['steps'] => raw .filter((s) => s.description && s.description.trim()) + .slice(0, Math.max(1, cap)) .map((s, i) => ({ id: `s${i + 1}`, description: s.description.trim(), @@ -18,21 +21,30 @@ const toSteps = ( suggestedTools: s.suggestedTools, })) +export interface PlanOutcome { + plan: IPlan + usage: IUsage + /** Names of configured skills the planner picked (validated). */ + skills: string[] +} + /** * Build the initial plan. Native mode uses generateObject(PlanSchema); prompted * mode salvages `{ reply, plan }` from plain text. Native failures (a weak model * that can't satisfy the schema) fall back to the prompted parse rather than * throwing, so the run degrades gracefully. An empty `steps` list signals the - * runner to answer directly (greeting / question / unclear). + * runner to answer directly (greeting / small talk). */ export const createPlan = async ( ctx: AgentContext, goal: string, /** Extra instruction appended to the goal (used by the empty-plan retry). */ nudge?: string, -): Promise<{ plan: IPlan; usage: IUsage }> => { +): Promise => { const state = await ctx.state() const effectiveGoal = nudge ? `${goal}\n\n${nudge}` : goal + const skillIndex = ctx.skills?.length ? renderSkillIndex(ctx.skills) : undefined + const cap = ctx.config.maxPlanSteps const commonFor = (mode: ToolCallMode) => { const parts = ctx.prompts.planner({ goal: effectiveGoal, @@ -40,26 +52,46 @@ export const createPlan = async ( toolCatalog: ctx.toolCatalog, mode, history: ctx.history, + skills: skillIndex, + searchMode: ctx.strategy === 'search', }) return { - system: systemFor(ctx, parts.system), - prompt: parts.prompt, + ...stageCall(ctx, 'planner', parts.system), + ...promptFor(ctx, parts.prompt), maxOutputTokens: ctx.config.budgets.planner, temperature: ctx.config.temperature, abortSignal: ctx.signal, timeoutMs: ctx.config.chatTimeoutMs, } } + const validSkills = (names: string[] | undefined): string[] => { + const known = new Set((ctx.skills ?? []).map((s) => s.name)) + const picked = [...new Set(names ?? [])] + const kept = picked.filter((n) => known.has(n)) + if (kept.length < picked.length) { + ctx.log.warn( + 'planner picked unknown skills, dropped:', + picked.filter((n) => !known.has(n)), + ) + } + return kept + } if (ctx.plannerMode === 'native') { try { const result = await generateStructured(ctx.plannerModel, PlanSchema, commonFor('native')) ctx.log.debug('planner (native):', result.object) return { - plan: { thought: result.object.thought, steps: toSteps(result.object.steps) }, + plan: { thought: result.object.thought, steps: toSteps(result.object.steps, cap) }, usage: normalizeUsage(result.usage), + skills: validSkills(result.object.skills), } } catch (err) { + // An abort is a stop, not a schema failure — don't spend a fallback call. + if (ctx.signal?.aborted) throw err + // A refused image is not a schema problem either: say so clearly. + const refused = imageRefusal(ctx, err, ctx.plannerModel) + if (refused) throw refused // Graceful degradation: fall back to the salvage parser instead of failing. ctx.log.warn('planner: native structured output failed, salvaging:', asMessage(err)) ctx.emit({ type: 'retry', phase: 'plan', attempt: 1, error: asMessage(err) }) @@ -69,16 +101,22 @@ export const createPlan = async ( // The prompted path — also the fallback after a native failure. Rendered with // mode 'prompted' so the model gets explicit JSON-shape instructions even // when the schema-constrained call just failed. - const result = await generate(ctx.plannerModel, commonFor('prompted')) + const result = await generate(ctx.plannerModel, commonFor('prompted')).catch((err: unknown) => { + throw imageRefusal(ctx, err, ctx.plannerModel) ?? err + }) const parsed = parsePlannerResponse(result.text) ctx.log.debug('planner (prompted) raw:', result.text) ctx.log.debug('planner (prompted) parsed:', parsed) return { plan: { thought: parsed.reply, - steps: toSteps(parsed.plan.map((description) => ({ description }))), + steps: toSteps( + parsed.plan.map((description) => ({ description })), + cap, + ), }, usage: normalizeUsage(result.usage), + skills: validSkills(parsed.skills), } } diff --git a/src/agent/replanner.ts b/src/agent/replanner.ts index 9930b02..104d661 100644 --- a/src/agent/replanner.ts +++ b/src/agent/replanner.ts @@ -2,9 +2,10 @@ import { generate, generateStructured } from '../llm/generate.js' import { normalizeUsage } from '../llm/util.js' import { parseReplannerResponse, type ReplanDecision } from '../parse.js' import type { ToolCallMode } from '../prompts.js' +import { renderActiveSkills } from '../skills.js' import type { IUsage } from './loop-types.js' import { emptyUsage } from './loop-types.js' -import { systemFor, type AgentContext } from './internal.js' +import { stageCall, type AgentContext } from './internal.js' import { ReplanSchema } from './schemas.js' export interface ReplanOutcome { @@ -28,10 +29,18 @@ export const decideReplan = async ( remaining: string[], ): Promise => { const state = await ctx.state() + const active = ctx.activeSkills?.() ?? [] const commonFor = (mode: ToolCallMode) => { - const parts = ctx.prompts.replanner({ goal, state, done, remaining, mode }) + const parts = ctx.prompts.replanner({ + goal, + state, + done, + remaining, + mode, + activeSkills: active.length ? renderActiveSkills(active) : undefined, + }) return { - system: systemFor(ctx, parts.system), + ...stageCall(ctx, 'replanner', parts.system), prompt: parts.prompt, maxOutputTokens: ctx.config.budgets.replanner, temperature: ctx.config.temperature, diff --git a/src/agent/runner.ts b/src/agent/runner.ts index 348fabb..65cd59c 100644 --- a/src/agent/runner.ts +++ b/src/agent/runner.ts @@ -1,16 +1,38 @@ -import type { LanguageModel } from 'ai' +import type { LanguageModel, ToolSet } from 'ai' +import { resolveCaching } from '../caching.js' import type { BrowserAgentConfig } from '../config.js' import { resolveConfig } from '../config.js' -import type { AgentEvent, AgentEventHandler, Phase } from '../events.js' -import { compressHistory } from '../memory/compress.js' +import type { AgentEvent, AgentEventHandler, UsagePhase } from '../events.js' +import { checkLimits, type LimitBreach } from '../limits.js' +import { clip } from '../llm/util.js' +import { + compactSteps, + compressHistory, + estimateMessagesTokens, + estimateTokens, +} from '../memory/compress.js' import type { StoredMessage } from '../memory/store.js' import { buildModelFromStage, resolveStage } from '../providers/registry.js' import { defaultPrompts, type Prompts } from '../prompts.js' +import { createSkillTools, defineSkill, SKILL_TOOL_NAMES, type Skill } from '../skills.js' +import { + createApprovalState, + type ToolApprovalMode, + TOOL_APPROVAL_MODES, +} from '../tools/approval.js' import { selectToolMode } from '../tools/mode.js' import { renderCatalog } from '../tools/prompted.js' +import { + createFindToolsTool, + FIND_TOOLS_NAME, + renderSearchCatalog, + toolCatalogOf, + type ToolCatalogEntry, +} from '../tools/search.js' import type { AgentToolSet } from '../tools/types.js' +import { createCallCounter, schemaHintOf, wrapToolsForRun } from '../tools/wrap.js' import { executeStep, replanWanted } from './executor.js' -import type { AgentContext } from './internal.js' +import type { AgentContext, EffectiveToolStrategy } from './internal.js' import { addUsage, emptyUsage, @@ -23,12 +45,35 @@ import { createPlan } from './planner.js' import { decideReplan } from './replanner.js' import { synthesizeAnswer } from './synthesizer.js' import { createLogger } from '../logger.js' +import { + AttachmentsNotSupportedError, + attachmentKind, + modelLabel, + unsupportedAttachment, + type AttachmentKind, + type RunFile, + type RunImage, +} from '../images.js' +import { supportsImages, supportsPdf } from '../providers/capabilities.js' export interface RunOptions { onEvent?: AgentEventHandler signal?: AbortSignal /** Override the config's sessionId for this run. */ sessionId?: string + /** + * Images sent with the goal (pasted screenshots, photos). They reach the + * planner, the executor and the synthesizer of a vision-capable model; a + * model known not to take images ends the run at once with + * ImagesNotSupportedError's message (see `capabilities.images`). + */ + images?: RunImage[] + /** + * Other attachments: PDFs and files the provider reads natively, as data or + * as an http(s) URL. Same rules as `images`, per kind (`capabilities.pdf`, + * `capabilities.files`). + */ + files?: RunFile[] } export interface RunResult { @@ -41,10 +86,50 @@ export interface RunResult { applied: number stopped: boolean usage: IUsage + /** Skills whose instructions were active in this run. */ + skills: string[] + /** The limit that stopped the run early, if any. */ + budgetExceeded?: LimitBreach +} + +export interface CompactOptions { + /** Session to compact (default: the config's sessionId). */ + sessionId?: string + signal?: AbortSignal + /** Compact even when below the threshold (default true for a manual call). */ + force?: boolean +} + +export interface CompactResult { + compacted: boolean + beforeTokens: number + afterTokens: number + usage: IUsage } export interface Agent { run(goal: string, opts?: RunOptions): Promise + /** Summarise the stored transcript (needs `memory`); returns the before/after size. */ + compact(opts?: CompactOptions): Promise + /** Switch the consent policy — e.g. the "autopilot" toggle. Applies to the next tool call. */ + setToolApprovalMode(mode: ToolApprovalMode): void + readonly toolApprovalMode: ToolApprovalMode + /** The tool catalogue (host + MCP tools after filtering). */ + listTools(): ToolCatalogEntry[] + /** The effective tool selection ('auto' resolved by catalogue size). */ + readonly toolStrategy: EffectiveToolStrategy + /** Configured skills (name + description). */ + readonly skills: { name: string; description: string }[] + /** + * What the executor model can take, per attachment kind: true / false, or + * undefined when unknown (the run tries, and a provider refusal becomes a + * clear error). + */ + readonly capabilities: { + images: boolean | undefined + pdf: boolean | undefined + files: boolean | undefined + } /** The resolved models, for hosts that want to reuse them (e.g. warm-up). */ readonly models: { planner: LanguageModel @@ -67,6 +152,19 @@ const filterTools = ( return Object.fromEntries(entries) } +const asJson = (v: unknown): string => { + if (typeof v === 'string') return v + try { + return JSON.stringify(v) ?? String(v) + } catch { + return String(v) + } +} + +// Budget of the tool-result excerpts the synthesizer answers from. +const FINDING_CHARS = 600 +const FINDINGS_BUDGET = 6_000 + /** * Create a headless plan → execute → replan → synthesize agent. Models are * resolved eagerly (dynamic provider imports + vault key fetch), so this is @@ -92,16 +190,70 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => const plannerMode = selectToolMode(plannerModel, cfg.toolMode) const executorMode = selectToolMode(executorModel, cfg.toolMode) - const tools: AgentToolSet = filterTools( + // Declaration order, never re-sorted: the host merges its tool sets in a + // fixed order (so the cached prefix is stable), and a server's own order is + // what small models were measured against (see the Node sibling's live gate). + const baseTools: AgentToolSet = filterTools( config.tools ?? {}, config.availableTools, config.excludedTools, ) - const toolCatalog = renderCatalog(tools) + + const skills: Skill[] = (config.skills ?? []).map(defineSkill) + const skillNames = new Set() + for (const s of skills) { + if (skillNames.has(s.name)) throw new Error(`duplicate skill name "${s.name}"`) + skillNames.add(s.name) + } + + const catalog = toolCatalogOf(baseTools) + const strategy: EffectiveToolStrategy = + cfg.toolSelectionStrategy === 'auto' + ? catalog.length > cfg.toolSearchThreshold + ? 'search' + : 'all' + : cfg.toolSelectionStrategy + + const reserved = [ + ...(skills.length ? SKILL_TOOL_NAMES : []), + ...(strategy === 'search' ? [FIND_TOOLS_NAME] : []), + ] + for (const name of reserved) { + if (Object.hasOwn(baseTools, name)) { + throw new Error( + `tool name "${name}" is reserved by the agent's built-in tools; rename your tool`, + ) + } + } + const builtinTools = new Set(reserved) + + // Parameter hints for prompted catalogues, derived once from the schemas. + const hints = new Map() + await Promise.all( + Object.entries(baseTools).map(async ([name, t]) => { + const hint = await schemaHintOf(t) + if (hint) hints.set(name, hint) + }), + ) + const toolHint = (name: string): string | undefined => hints.get(name) + + const plannerCatalog = + strategy === 'search' ? renderSearchCatalog(catalog) : renderCatalog(baseTools, toolHint) const prompts: Prompts = { ...defaultPrompts, ...config.prompts } + const caching = resolveCaching(config.promptCaching) + // The prompted path is text-only; otherwise trust the host's `vision`, then + // what is known about the model. + const textOnly = executorMode === 'prompted' + const accepts: Record = { + image: + config.inputs?.images ?? config.vision ?? (textOnly ? false : supportsImages(executorModel)), + pdf: config.inputs?.pdf ?? (textOnly ? false : supportsPdf(executorModel)), + file: config.inputs?.files ?? (textOnly ? false : undefined), + } + const approval = createApprovalState(config.toolApproval) const log = createLogger(cfg.logLevel, config.logger) log.info( - `agent ready — tools: [${Object.keys(tools).join(', ') || 'none'}], planner mode: ${plannerMode}, executor mode: ${executorMode}`, + `agent ready — ${catalog.length} tool(s) (${strategy}), skills: [${[...skillNames].join(', ') || 'none'}], planner mode: ${plannerMode}, executor mode: ${executorMode}, approval: ${approval.mode}`, ) const run = async (goal: string, opts: RunOptions = {}): Promise => { @@ -124,13 +276,14 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => applied: 0, stopped: false, usage, + skills: [], } if (!text) return result const state = async (): Promise => config.describeState ? config.describeState() : undefined const isAborted = (): boolean => opts.signal?.aborted === true - const bumpUsage = (u: IUsage, phase: Phase): void => { + const bumpUsage = (u: IUsage, phase: UsagePhase): void => { usage = addUsage(usage, u) result.usage = usage emit({ type: 'usage', phase, usage: u }) @@ -144,7 +297,76 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => } } + // ── per-run tool state: skills, discovered tools, call budget, current step + const activeSkillNames = new Set() + const activateSkill = (name: string, by: 'plan' | 'tool'): void => { + if (activeSkillNames.has(name) || !skillNames.has(name)) return + activeSkillNames.add(name) + result.skills = [...activeSkillNames] + emit({ type: 'skill.activated', name, by }) + } + const discovered: string[] = [] + let currentStep: IPlanStep | undefined + const counter = createCallCounter(cfg.maxToolCalls) + const builtins: ToolSet = {} + if (skills.length) { + Object.assign( + builtins, + createSkillTools(skills, (s) => activateSkill(s.name, 'tool')), + ) + } + if (strategy === 'search') { + builtins[FIND_TOOLS_NAME] = createFindToolsTool( + () => catalog, + (query, names) => { + for (const n of names) { + const i = discovered.indexOf(n) + if (i >= 0) discovered.splice(i, 1) + discovered.push(n) + } + emit({ type: 'tools.discovered', step: currentStep, query, names }) + }, + toolHint, + ) + } + const tools = wrapToolsForRun( + { ...baseTools, ...builtins }, + { + approval, + builtins: builtinTools, + emit, + step: () => currentStep, + signal: opts.signal, + addUsage: (u) => bumpUsage(u, 'subagent'), + countCall: counter.count, + maxToolOutputChars: cfg.compaction.maxToolOutputChars, + }, + ) + + let budgetExceeded: LimitBreach | undefined + const breach = (extra?: IUsage): LimitBreach | undefined => { + const b = checkLimits(extra ? addUsage(usage, extra) : usage, cfg.limits) + if (b) return b + if (counter.exhausted) + return { kind: 'tool-calls', tokens: counter.calls, cap: cfg.maxToolCalls } + return undefined + } + emit({ type: 'run.start', goal: text }) + const images = [...(opts.images ?? []), ...(opts.files ?? [])] + const refused = images.map(attachmentKind).find((kind) => accepts[kind] === false) + if (refused) { + // Fail before spending a single token, with a message a user can act on. + const err = unsupportedAttachment( + modelLabel(executorModel), + refused, + textOnly ? 'it runs in the text-only prompted tool mode' : undefined, + ) + log.warn(err.message) + emit({ type: 'error', phase: 'run', error: err.message }) + result.final = err.message + return result + } // Read back the session transcript BEFORE appending the current goal, so // the planner can resolve references to earlier turns ("make it bigger"). let history: StoredMessage[] = [] @@ -155,6 +377,33 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => /* persistence is best-effort */ } } + // Auto-compaction of a long transcript BEFORE planning, so the run starts + // within the window (configured `compaction` only; the legacy + // compressAfterChars path compacts after the run, as it always did). + if (config.memory && config.compaction && cfg.compaction.auto && history.length > 0) { + const before = estimateMessagesTokens(history) + if (before > cfg.compaction.thresholdTokens) { + const compacted = await compressHistory(history, synthesizerModel, { + thresholdTokens: cfg.compaction.thresholdTokens, + keepRecent: cfg.compaction.keepRecentTurns, + summaryMaxTokens: cfg.budgets.compaction, + timeoutMs: cfg.chatTimeoutMs, + abortSignal: opts.signal, + onUsage: (u) => bumpUsage(u, 'compact'), + }) + if (compacted !== history) { + history = compacted + await config.memory.replace(sessionId, compacted).catch(() => {}) + emit({ + type: 'context.compacted', + scope: 'history', + beforeTokens: before, + afterTokens: estimateMessagesTokens(compacted), + }) + } + } + } + const ctx: AgentContext = { config: cfg, raw: config, @@ -164,26 +413,47 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => plannerMode, executorMode, tools, - toolCatalog, + toolCatalog: plannerCatalog, prompts, emit, log, signal: opts.signal, state, history, + strategy, + caching, + skills, + activeSkills: () => skills.filter((s) => activeSkillNames.has(s.name)), + discovered: () => discovered, + builtinTools, + toolHint, + usageSoFar: () => usage, + overBudget: (extra) => breach(extra) !== undefined, + setCurrentStep: (step) => { + currentStep = step + }, + images: images.length ? images : undefined, + } + // Memory is text: an image is remembered by its name, not its pixels. + await remember({ + role: 'user', + content: images.length + ? `${text}\n[attached: ${images.map((f, i) => f.name ?? `${attachmentKind(f)} ${i + 1}`).join(', ')}]` + : text, + }) + log.info('run:', text, images.length ? `(+${images.length} image(s))` : '') + + const stop = (): RunResult => { + emit({ type: 'stopped' }) + result.stopped = true + return result } - await remember({ role: 'user', content: text }) - log.info('run:', text) try { // 1) PLAN let planned = await createPlan(ctx, text) bumpUsage(planned.usage, 'plan') - if (isAborted()) { - emit({ type: 'stopped' }) - result.stopped = true - return result - } + if (isAborted()) return stop() // Small models sometimes return an empty plan for a clearly actionable // goal — and put a hallucinated "done!" into the thought, so the run @@ -197,19 +467,15 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => const retried = await createPlan( ctx, text, - 'NOTE: If the message above asks to build, add, change, clear or delete ANYTHING, you MUST output 1-6 concrete steps. Output an empty plan ONLY for a pure greeting or question.', + 'NOTE: If the message above asks to build, add, change, clear, delete, find, check or look up ANYTHING the tools can do, you MUST output 1-6 concrete steps. Output an empty plan ONLY for a pure greeting or small talk.', ) bumpUsage(retried.usage, 'plan') if (retried.plan.steps.length > 0) planned = retried } result.plan = planned.plan - if (isAborted()) { - emit({ type: 'stopped' }) - result.stopped = true - return result - } + if (isAborted()) return stop() - // Greeting / question / unclear → answer directly, run no tools. + // Greeting / small talk → answer directly, run no tools. if (planned.plan.steps.length === 0) { log.info('no plan — answering directly (no tools will run)') const answer = planned.plan.thought.trim() || "Tell me what you'd like to do." @@ -224,21 +490,27 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => ) emit({ type: 'plan.created', plan: planned.plan }) planned.plan.steps.forEach((step, index) => emit({ type: 'plan.step-added', step, index })) + for (const name of planned.skills) activateSkill(name, 'plan') // 2) EXECUTE → REPLAN - const done: string[] = [] + let done: string[] = [] + const findings: string[] = [] let remaining: IPlanStep[] = [...planned.plan.steps] let iter = 0 let revisions = 0 while (remaining.length > 0 && iter < cfg.maxIterations) { - if (isAborted()) { - emit({ type: 'stopped' }) - result.stopped = true - return result + if (isAborted()) return stop() + const hit = breach() + if (hit) { + budgetExceeded = hit + result.budgetExceeded = hit + log.warn(`${hit.kind} limit reached (${hit.tokens}/${hit.cap}) — writing the answer now`) + emit({ type: 'budget.exceeded', ...hit }) + break } iter += 1 const step = remaining.shift() as IPlanStep - const stepNo = done.length + 1 + const stepNo = result.trace.length + 1 const total = stepNo + remaining.length emit({ type: 'step.start', step, index: stepNo, total }) @@ -250,13 +522,13 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => } catch (err) { // A user abort surfaces as a thrown AbortError — that is a stop, not // a step failure. - if (isAborted()) { - emit({ type: 'stopped' }) - result.stopped = true - return result - } + if (isAborted()) return stop() + // A model that refuses the attachments can't do any step: end the run. + if (err instanceof AttachmentsNotSupportedError) throw err emit({ type: 'error', phase: 'execute', error: errMessage(err) }) stepResult = { step, summary: errMessage(err), toolCalls: [], blocked: true } + } finally { + currentStep = undefined } result.trace.push(stepResult) @@ -275,9 +547,43 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => .forEach((c) => log.warn(`tool ${c.name} failed:`, c.output)) log.debug(`step ${stepNo} detail:`, stepResult) emit({ type: 'step.complete', step, result: stepResult }) + // The step's own summary carries the data later steps and the answer + // need ("found 3 open issues: #12, #15, #19") — not just a count. + const counts = stepResult.toolCalls.length + ? ` [${applied} tool call(s) ok${failed ? `, ${failed} failed` : ''}]` + : '' done.push( - `${step.description} — ${applied} applied${failed ? `, ${failed} failed` : ''}${stepResult.blocked ? ', blocked' : ''}`, + `${step.description} — ${clip(stepResult.summary, FINDING_CHARS)}${counts}${stepResult.blocked ? ' [blocked]' : ''}`, ) + for (const c of stepResult.toolCalls) { + if (!c.ok || builtinTools.has(c.name)) continue + findings.push(`- ${c.name}: ${clip(asJson(c.output), FINDING_CHARS)}`) + } + while (findings.join('\n').length > FINDINGS_BUDGET && findings.length > 1) findings.shift() + + // Auto-compaction of the run's step log when it outgrows the threshold. + if (cfg.compaction.auto) { + const before = estimateTokens(done.join('\n')) + if (before > cfg.compaction.thresholdTokens) { + const compacted = await compactSteps(done, synthesizerModel, { + thresholdTokens: cfg.compaction.thresholdTokens, + keepRecent: cfg.compaction.keepRecentSteps, + summaryMaxTokens: cfg.budgets.compaction, + timeoutMs: cfg.chatTimeoutMs, + abortSignal: opts.signal, + onUsage: (u) => bumpUsage(u, 'compact'), + }) + if (compacted !== done) { + done = compacted + emit({ + type: 'context.compacted', + scope: 'trace', + beforeTokens: before, + afterTokens: estimateTokens(done.join('\n')), + }) + } + } + } // 2b) REPLAN — by default after a blocked / failed step; `replanAfter` // can widen the trigger ('always', or a host predicate reacting to @@ -290,6 +596,7 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => iter < cfg.maxIterations && revisions < cfg.maxRevisions && !isAborted() && + !breach() && (await replanWanted(cfg.replanAfter, stepResult, { signal: ctx.signal, timeoutMs: cfg.chatTimeoutMs, @@ -308,7 +615,7 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => if (decision.decision === 'revise' && decision.plan.length > 0) { revisions += 1 remaining = decision.plan - .slice(0, cfg.maxIterations - iter) + .slice(0, Math.min(cfg.maxPlanSteps, cfg.maxIterations - iter)) .map((description, i) => ({ id: `r${iter}-${i + 1}`, description })) emit({ type: 'plan.revised', @@ -318,21 +625,28 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => } } } - - if (isAborted()) { - emit({ type: 'stopped' }) - result.stopped = true - return result + // A limit reached on the very last step is still worth reporting. + if (!budgetExceeded && remaining.length > 0) { + const hit = breach() + if (hit) { + budgetExceeded = hit + result.budgetExceeded = hit + emit({ type: 'budget.exceeded', ...hit }) + } } + if (isAborted()) return stop() + // 3) SYNTHESIZE let summary = 'Done — the changes have been applied.' if (cfg.synthesize) { try { - const synth = await synthesizeAnswer(ctx, text, done) + const synth = await synthesizeAnswer(ctx, text, done, findings) bumpUsage(synth.usage, 'synthesize') if (synth.text) summary = synth.text - } catch { + } catch (err) { + if (isAborted()) return stop() + if (err instanceof AttachmentsNotSupportedError) throw err /* keep the default */ } } @@ -341,30 +655,92 @@ export const createAgent = async (config: BrowserAgentConfig): Promise => await remember({ role: 'assistant', content: summary }) result.final = summary - // 4) COMPRESS persisted history - if (config.memory && cfg.compressAfterChars > 0) { + // 4) COMPACT persisted history (legacy chars threshold, or the configured + // compaction's token threshold). + if (config.memory && (config.compaction ? cfg.compaction.auto : cfg.compressAfterChars > 0)) { try { const transcript = await config.memory.load(sessionId) + const before = estimateMessagesTokens(transcript) const compacted = await compressHistory(transcript, synthesizerModel, { - maxChars: cfg.compressAfterChars, + ...(config.compaction + ? { + thresholdTokens: cfg.compaction.thresholdTokens, + keepRecent: cfg.compaction.keepRecentTurns, + summaryMaxTokens: cfg.budgets.compaction, + } + : { maxChars: cfg.compressAfterChars }), timeoutMs: cfg.chatTimeoutMs, abortSignal: opts.signal, + onUsage: (u) => bumpUsage(u, 'compact'), }) - if (compacted !== transcript) await config.memory.replace(sessionId, compacted) + if (compacted !== transcript) { + await config.memory.replace(sessionId, compacted) + emit({ + type: 'context.compacted', + scope: 'history', + beforeTokens: before, + afterTokens: estimateMessagesTokens(compacted), + }) + } } catch { /* best-effort */ } } return result } catch (err) { + if (isAborted()) return stop() emit({ type: 'error', phase: 'run', error: errMessage(err) }) result.final = errMessage(err) return result } } + const compact = async (o: CompactOptions = {}): Promise => { + const empty: CompactResult = { + compacted: false, + beforeTokens: 0, + afterTokens: 0, + usage: emptyUsage(), + } + if (!config.memory) return empty + const sessionId = o.sessionId ?? cfg.sessionId + const transcript = await config.memory.load(sessionId) + const beforeTokens = estimateMessagesTokens(transcript) + let usage = emptyUsage() + const compacted = await compressHistory(transcript, synthesizerModel, { + thresholdTokens: cfg.compaction.thresholdTokens, + keepRecent: cfg.compaction.keepRecentTurns, + summaryMaxTokens: cfg.budgets.compaction, + force: o.force ?? true, + timeoutMs: cfg.chatTimeoutMs, + abortSignal: o.signal, + onUsage: (u) => { + usage = addUsage(usage, u) + }, + }) + if (compacted === transcript) + return { ...empty, beforeTokens, afterTokens: beforeTokens, usage } + await config.memory.replace(sessionId, compacted) + return { compacted: true, beforeTokens, afterTokens: estimateMessagesTokens(compacted), usage } + } + return { run, + compact, + setToolApprovalMode: (mode: ToolApprovalMode) => { + if (!TOOL_APPROVAL_MODES.includes(mode)) { + throw new Error(`unknown tool approval mode "${mode}" (${TOOL_APPROVAL_MODES.join(' | ')})`) + } + approval.mode = mode + log.info(`tool approval mode → ${mode}`) + }, + get toolApprovalMode() { + return approval.mode + }, + listTools: () => catalog.map((e) => ({ ...e })), + toolStrategy: strategy, + skills: skills.map((s) => ({ name: s.name, description: s.description })), + capabilities: { images: accepts.image, pdf: accepts.pdf, files: accepts.file }, models: { planner: plannerModel, executor: executorModel, synthesizer: synthesizerModel }, } } diff --git a/src/agent/schemas.ts b/src/agent/schemas.ts index 6d44d91..0152579 100644 --- a/src/agent/schemas.ts +++ b/src/agent/schemas.ts @@ -12,9 +12,11 @@ export const PlanStepSchema = z.object({ }) export const PlanSchema = z.object({ - /** One short sentence. For a greeting/question/unclear request, holds the answer and steps is empty. */ + /** One short sentence. For a greeting/small talk, holds the answer and steps is empty. */ thought: z.string(), steps: z.array(PlanStepSchema), + /** Names of the configured skills that apply to this goal. */ + skills: z.array(z.string()).optional(), }) export const ReplanSchema = z.object({ diff --git a/src/agent/synthesizer.ts b/src/agent/synthesizer.ts index b9028b9..5a23ad5 100644 --- a/src/agent/synthesizer.ts +++ b/src/agent/synthesizer.ts @@ -1,26 +1,36 @@ import { stream } from '../llm/generate.js' import { normalizeUsage } from '../llm/util.js' import { looksLikeJson, parsePlainText } from '../parse.js' +import { renderActiveSkills } from '../skills.js' import type { IUsage } from './loop-types.js' -import { systemFor, type AgentContext } from './internal.js' +import { imageRefusal, promptFor, stageCall, type AgentContext } from './internal.js' /** - * Write the final natural-language summary of what the run accomplished, - * streaming it as `final.text-delta` events. Plain text only (parsePlainText - * strips any stray JSON; deltas are suppressed entirely when the model drifts - * into JSON, so raw structure never reaches a UI). Falls back to a default - * sentence if the model returns nothing usable. + * Write the final natural-language answer of the run, streaming it as + * `final.text-delta` events (and thoughts as `final.reasoning-delta`). Plain + * text only (parsePlainText strips any stray JSON; deltas are suppressed + * entirely when the model drifts into JSON, so raw structure never reaches a + * UI). Falls back to a default sentence if the model returns nothing usable. */ export const synthesizeAnswer = async ( ctx: AgentContext, goal: string, done: string[], + /** Excerpts of what the tools returned — the data the answer is made of. */ + findings: string[] = [], ): Promise<{ text: string; usage: IUsage }> => { const state = await ctx.state() - const parts = ctx.prompts.synthesizer({ goal, state, done }) + const active = ctx.activeSkills?.() ?? [] + const parts = ctx.prompts.synthesizer({ + goal, + state, + done, + findings, + activeSkills: active.length ? renderActiveSkills(active) : undefined, + }) const result = stream(ctx.synthesizerModel, { - system: systemFor(ctx, parts.system), - prompt: parts.prompt, + ...stageCall(ctx, 'synthesizer', parts.system), + ...promptFor(ctx, parts.prompt), maxOutputTokens: ctx.config.budgets.synthesizer, temperature: ctx.config.temperature, abortSignal: ctx.signal, @@ -29,13 +39,30 @@ export const synthesizeAnswer = async ( let text = '' let verdict: 'unknown' | 'emit' | 'suppress' = 'unknown' - for await (const delta of result.textStream) { + for await (const part of result.fullStream) { + if (part.type === 'reasoning-delta') { + if (part.text) ctx.emit({ type: 'final.reasoning-delta', delta: part.text }) + continue + } + if (part.type === 'error') { + const err = part.error instanceof Error ? part.error : new Error(String(part.error)) + throw imageRefusal(ctx, err, ctx.synthesizerModel) ?? err + } + if (part.type !== 'text-delta') continue + const delta = part.text text += delta if (verdict === 'unknown') { const lead = text.trimStart() if (!lead) continue + // Inline blocks (local reasoning models) are not part of the answer. + if (lead.startsWith('') && !text.includes('')) continue verdict = looksLikeJson(lead) ? 'suppress' : 'emit' - if (verdict === 'emit') ctx.emit({ type: 'final.text-delta', delta: text }) + if (verdict === 'emit') { + const visible = text.includes('') + ? text.slice(text.lastIndexOf('') + ''.length).trimStart() + : text + if (visible) ctx.emit({ type: 'final.text-delta', delta: visible }) + } } else if (verdict === 'emit') { ctx.emit({ type: 'final.text-delta', delta }) } diff --git a/src/caching.ts b/src/caching.ts new file mode 100644 index 0000000..4f1531d --- /dev/null +++ b/src/caching.ts @@ -0,0 +1,94 @@ +import type { SystemModelMessage } from 'ai' +import type { ProviderOptionsMap } from './thinking.js' + +/** + * Prompt caching. Every stage's SYSTEM prompt holds only run-stable content + * (role, domain context, skills, tool catalogue) and the dynamic parts go to + * the user prompt — so OpenAI's and Gemini's automatic prefix caches hit. + * Anthropic caches only behind an explicit breakpoint, placed on the system + * message; OpenAI additionally routes by `promptCacheKey`. Other providers + * ignore option keys that are not theirs. + */ +export type PromptCachingSetting = boolean | { ttl?: '5m' | '1h'; key?: string } + +export interface ResolvedCaching { + enabled: boolean + ttl?: '5m' | '1h' + key?: string +} + +export const resolveCaching = (s: PromptCachingSetting | undefined): ResolvedCaching => { + if (s === false) return { enabled: false } + if (s === undefined || s === true) return { enabled: true } + return { enabled: true, ttl: s.ttl, key: s.key } +} + +/** The system prompt as the AI SDK `instructions`, with a cache breakpoint when enabled. */ +export const cachedInstructions = ( + system: string, + caching: ResolvedCaching, +): string | SystemModelMessage => + caching.enabled + ? { + role: 'system', + content: system, + providerOptions: { + anthropic: { + cacheControl: { type: 'ephemeral', ...(caching.ttl ? { ttl: caching.ttl } : {}) }, + }, + }, + } + : system + +/** Call-level provider options for caching (OpenAI cache routing key). */ +export const cachingProviderOptions = ( + caching: ResolvedCaching, + clientName: string, + stage: string, +): ProviderOptionsMap | undefined => + caching.enabled + ? { openai: { promptCacheKey: caching.key ?? `${clientName}:${stage}` } } + : undefined + +/** An Anthropic cache breakpoint (other providers ignore the key). */ +const breakpoint = (ttl?: '5m' | '1h') => ({ + anthropic: { cacheControl: { type: 'ephemeral' as const, ...(ttl ? { ttl } : {}) } }, +}) + +/** The message without an Anthropic breakpoint (other provider options kept). */ +const withoutBreakpoint = (m: M): M => { + const own = m.providerOptions as Record> | undefined + if (!own?.anthropic || !('cacheControl' in own.anthropic)) return m + const { cacheControl: _drop, ...anthropic } = own.anthropic + const { anthropic: _old, ...rest } = own + const providerOptions = Object.keys(anthropic).length ? { ...rest, anthropic } : rest + const { providerOptions: _po, ...base } = m as M & { providerOptions?: unknown } + return (Object.keys(providerOptions).length ? { ...base, providerOptions } : base) as M +} + +/** + * The conversation with a rolling cache breakpoint on its LAST message — the + * agent-loop pattern Claude Code uses: every tool-calling round re-sends the + * rounds before it, and with the breakpoint moved to the newest message each + * request reads all of them from the cache and writes only the new tail. + * Earlier message breakpoints are removed (the SDK carries a round's messages + * into the next one), so a request holds at most two: system + newest — + * Anthropic rejects more than four. + */ +export const withRollingBreakpoint = ( + messages: M[], + caching: ResolvedCaching, +): M[] => { + if (!caching.enabled || messages.length === 0) return messages + const out = messages.map((m, i) => (i < messages.length - 1 ? withoutBreakpoint(m) : m)) + const last = out[out.length - 1] + const own = (last.providerOptions ?? {}) as Record> + out[out.length - 1] = { + ...last, + providerOptions: { + ...own, + anthropic: { ...own.anthropic, ...breakpoint(caching.ttl).anthropic }, + }, + } + return out +} diff --git a/src/config.ts b/src/config.ts index 3614892..0c1d294 100644 --- a/src/config.ts +++ b/src/config.ts @@ -1,9 +1,19 @@ import type { IStepResult } from './agent/loop-types.js' +import type { PromptCachingSetting } from './caching.js' +import type { TokenLimits } from './limits.js' +import { + resolveCompaction, + type CompactionConfig, + type ResolvedCompaction, +} from './memory/compress.js' import type { ContextStore } from './memory/store.js' import type { AgentLoggerSink, LogLevel } from './logger.js' import type { ModelInput, StageInput } from './providers/types.js' import type { Prompts } from './prompts.js' import type { CredentialStore } from './secrets/store.js' +import type { Skill } from './skills.js' +import type { ThinkingSetting, ThinkingStage } from './thinking.js' +import type { ToolApprovalConfig } from './tools/approval.js' import type { AgentToolSet } from './tools/types.js' /** Force a tool-calling strategy, or let the agent pick per model ('auto'). */ @@ -24,8 +34,12 @@ export type ReplanTrigger = * 'all' — the executor sees the full ToolSet on every step. * 'plan-narrowed' — the executor sees only a step's suggestedTools (the planner * must populate them). Empty means a reasoning-only step. + * 'search' — for large catalogues: the executor starts with the step's + * suggestedTools plus a `find_tools` meta-tool that activates + * more on demand; the planner sees a condensed catalogue. + * 'auto' — 'all' up to `toolSearchThreshold` tools, 'search' above (default). */ -export type ToolSelectionStrategy = 'all' | 'plan-narrowed' +export type ToolSelectionStrategy = 'all' | 'plan-narrowed' | 'search' | 'auto' /** Per-phase generation budgets (max output tokens). */ export interface PhaseBudgets { @@ -54,7 +68,15 @@ export interface BrowserAgentConfig { excludedTools?: string[] /** Force native/prompted tool-calling; 'auto' picks per model (default 'auto'). */ toolMode?: ToolMode + /** How the executor sees the tool catalogue (default 'auto'). */ toolSelectionStrategy?: ToolSelectionStrategy + /** Catalogue size above which 'auto' switches to 'search' (default 40). */ + toolSearchThreshold?: number + /** Consent policy for tool calls: autopilot, ask before writes, ask always, read-only. */ + toolApproval?: ToolApprovalConfig + + /** Reusable instruction bundles (SKILL.md) the planner/executor can pull in. */ + skills?: Skill[] /** Prepended to every phase's system prompt. */ systemPrompt?: string @@ -77,18 +99,45 @@ export interface BrowserAgentConfig { maxStepsPerTask?: number /** Cap on replanner "revise" decisions per run (default 2). */ maxRevisions?: number + /** Cap on tool calls per run (default: unlimited). */ + maxToolCalls?: number + /** Cap on the steps of one plan (default 8). */ + maxPlanSteps?: number /** Per-call timeout so a hang can't freeze a run (default 120000). */ chatTimeoutMs?: number + /** Legacy per-phase output caps; `limits.perCall` wins when both are set. */ budgets?: PhaseBudgets + /** Run-level token budgets (input / output / thinking / total) + per-call output caps. */ + limits?: TokenLimits temperature?: number + /** Thinking for every stage: true, a level ('low'…'xhigh'), or { level, budgetTokens }. */ + thinking?: ThinkingSetting + /** Per-stage thinking; a stage entry wins over `thinking`. */ + stageThinking?: Partial> + /** + * Whether the model accepts images. Default: inferred (`supportsImages`) — + * local/prompted models are text-only; unknown models are tried and a + * refusal is reported as ImagesNotSupportedError. + */ + vision?: boolean + /** Per-kind override of what the model takes as attachments (`images` wins over `vision`). */ + inputs?: { images?: boolean; pdf?: boolean; files?: boolean } + /** Provider prompt caching (default true): stable system prefixes + Anthropic breakpoints. */ + promptCaching?: PromptCachingSetting + /** Context compaction (auto + manual `agent.compact()`); see CompactionConfig. */ + compaction?: CompactionConfig + /** Master switch for the replan phase (default true). */ replan?: boolean /** What triggers the replanner when it is on (default 'failure'). */ replanAfter?: ReplanTrigger /** Write a final natural-language summary (default true). */ synthesize?: boolean - /** Compress stored history past this many chars (0 = off, default 12000). */ + /** + * Legacy: compress stored history past this many chars after each run + * (0 = off). Used only when `compaction` is not set. + */ compressAfterChars?: number /** Verbosity: 'silent' | 'error' | 'warn' (default) | 'info' | 'debug'. */ @@ -102,16 +151,22 @@ export interface ResolvedConfig { sessionId: string toolMode: ToolMode toolSelectionStrategy: ToolSelectionStrategy + toolSearchThreshold: number maxIterations: number maxStepsPerTask: number maxRevisions: number + maxToolCalls: number + maxPlanSteps: number chatTimeoutMs: number - budgets: Required + /** Effective per-call output caps (limits.perCall over budgets). */ + budgets: Required & { compaction: number } + limits: TokenLimits temperature: number | undefined replan: boolean replanAfter: ReplanTrigger synthesize: boolean compressAfterChars: number + compaction: ResolvedCompaction logLevel: LogLevel } @@ -119,21 +174,27 @@ export const resolveConfig = (c: BrowserAgentConfig): ResolvedConfig => ({ clientName: c.clientName ?? 'agent-web', sessionId: c.sessionId ?? 'default', toolMode: c.toolMode ?? 'auto', - toolSelectionStrategy: c.toolSelectionStrategy ?? 'all', + toolSelectionStrategy: c.toolSelectionStrategy ?? 'auto', + toolSearchThreshold: c.toolSearchThreshold ?? 40, maxIterations: c.maxIterations ?? 8, maxStepsPerTask: c.maxStepsPerTask ?? 4, maxRevisions: c.maxRevisions ?? 2, + maxToolCalls: c.maxToolCalls ?? 0, + maxPlanSteps: c.maxPlanSteps ?? 8, chatTimeoutMs: c.chatTimeoutMs ?? 120_000, budgets: { - planner: c.budgets?.planner ?? 800, - executor: c.budgets?.executor ?? 1200, - replanner: c.budgets?.replanner ?? 400, - synthesizer: c.budgets?.synthesizer ?? 400, + planner: c.limits?.perCall?.planner ?? c.budgets?.planner ?? 800, + executor: c.limits?.perCall?.executor ?? c.budgets?.executor ?? 1200, + replanner: c.limits?.perCall?.replanner ?? c.budgets?.replanner ?? 400, + synthesizer: c.limits?.perCall?.synthesizer ?? c.budgets?.synthesizer ?? 400, + compaction: c.limits?.perCall?.compaction ?? c.compaction?.summaryMaxTokens ?? 1024, }, + limits: c.limits ?? {}, temperature: c.temperature, replan: c.replan ?? true, replanAfter: c.replanAfter ?? 'failure', synthesize: c.synthesize ?? true, compressAfterChars: c.compressAfterChars ?? 12_000, + compaction: resolveCompaction(c.compaction), logLevel: c.logLevel ?? 'warn', }) diff --git a/src/events.ts b/src/events.ts index 1564df0..1501a9b 100644 --- a/src/events.ts +++ b/src/events.ts @@ -1,7 +1,10 @@ import type { IPlan, IPlanStep, IStepResult, IUsage } from './agent/loop-types.js' +import type { LimitKind } from './limits.js' export type ReplanMode = 'continue' | 'revise' | 'finish' export type Phase = 'plan' | 'execute' | 'replan' | 'synthesize' +/** Phases a `usage` event can be charged to: the four stages plus compaction and subagents. */ +export type UsagePhase = Phase | 'compact' | 'subagent' /** * Everything the agent does is streamed as a typed event so any UI (or none) @@ -19,14 +22,55 @@ export type AgentEvent = | { type: 'plan.revised'; plan: IPlan; reason: string } | { type: 'step.start'; step: IPlanStep; index: number; total: number } | { type: 'step.text-delta'; step: IPlanStep; delta: string } + /** The executor model's streamed thoughts (thinking enabled + provider support). */ + | { type: 'step.reasoning-delta'; step: IPlanStep; delta: string } | { type: 'step.tool-call'; step: IPlanStep; name: string; input: unknown } | { type: 'step.tool-result'; step: IPlanStep; name: string; output: unknown; ok: boolean } | { type: 'step.complete'; step: IPlanStep; result: IStepResult } | { type: 'replan.decision'; mode: ReplanMode; reason: string } | { type: 'final.text-delta'; delta: string } + /** The synthesizer's streamed thoughts. */ + | { type: 'final.reasoning-delta'; delta: string } | { type: 'final'; text: string } - | { type: 'usage'; phase: Phase; usage: IUsage } + | { type: 'usage'; phase: UsagePhase; usage: IUsage } | { type: 'retry'; phase: Phase; attempt: number; error: string } + /** A run-level limit was reached; the run stops executing and writes its answer. */ + | { type: 'budget.exceeded'; kind: LimitKind; tokens: number; cap: number } + /** Older context was summarised to fit the window. */ + | { + type: 'context.compacted' + /** history = stored transcript; trace = the run's done steps; tool-results = stale results inside a step's tool loop. */ + scope: 'history' | 'trace' | 'tool-results' + beforeTokens: number + afterTokens: number + } + /** A skill's instructions entered the context (picked by the planner, or loaded by a tool call). */ + | { type: 'skill.activated'; name: string; by: 'plan' | 'tool' } + /** find_tools activated more tools (large-catalogue search mode). */ + | { type: 'tools.discovered'; step?: IPlanStep; query: string; names: string[] } + /** A tool call is waiting for the user's consent (see `toolApproval`). */ + | { + type: 'tool.approval-requested' + id: string + name: string + input: unknown + readOnly: boolean + step?: IPlanStep + } + /** A consent decision: from the user, or `automatic` (policy denial). */ + | { + type: 'tool.approval-resolved' + id: string + name: string + approved: boolean + reason?: string + automatic: boolean + } + | { type: 'subagent.start'; id: string; name: string; task: string } + /** An event of a running subagent, forwarded verbatim. */ + | { type: 'subagent.event'; id: string; name: string; event: AgentEvent } + | { type: 'subagent.complete'; id: string; name: string; text: string; usage: IUsage } + | { type: 'subagent.error'; id: string; name: string; error: string } | { type: 'stopped' } | { type: 'error'; phase: Phase | 'run'; error: string } diff --git a/src/files/tools.ts b/src/files/tools.ts new file mode 100644 index 0000000..1f7d278 --- /dev/null +++ b/src/files/tools.ts @@ -0,0 +1,96 @@ +import { jsonSchema, tool, type ToolSet } from 'ai' +import { markReadOnly } from '../tools/approval.js' +import { isTextMime, type VirtualFileSystem } from './vfs.js' + +export interface FileToolsOptions { + /** Tool-name prefix (default 'fs_' → fs_list, fs_read, fs_write, fs_delete). */ + prefix?: string + /** Mount only the read tools (default false). */ + readOnly?: boolean + /** Cap on the text fs_read returns (default 50 000 chars; the rest is cut with a marker). */ + maxReadChars?: number +} + +/** + * Tools that give the agent the virtual file system: list, read, write and + * delete. Listing and reading are marked read-only, so the `ask-writes` + * consent mode lets them through and asks before writes and deletes. + */ +export const createFileTools = (vfs: VirtualFileSystem, opts: FileToolsOptions = {}): ToolSet => { + const p = opts.prefix ?? 'fs_' + const maxRead = opts.maxReadChars ?? 50_000 + const tools: ToolSet = { + [`${p}list`]: markReadOnly( + tool({ + description: + 'List the files in the workspace (attachments the user added, files you wrote). Optionally only under a directory prefix.', + inputSchema: jsonSchema<{ prefix?: string }>({ + type: 'object', + properties: { prefix: { type: 'string', description: 'Directory prefix, e.g. "/docs"' } }, + additionalProperties: false, + }), + execute: async ({ prefix }) => + (await vfs.list(prefix || '/')).map((f) => ({ + path: f.path, + mimeType: f.mimeType, + size: f.size, + })), + }), + ), + [`${p}read`]: markReadOnly( + tool({ + description: 'Read a text file from the workspace by its path.', + inputSchema: jsonSchema<{ path: string }>({ + type: 'object', + properties: { path: { type: 'string', description: 'Absolute path, e.g. "/notes.md"' } }, + required: ['path'], + additionalProperties: false, + }), + execute: async ({ path }) => { + const f = await vfs.read(path) + if (!f) { + const known = (await vfs.list()).map((x) => x.path).slice(0, 30) + throw new Error(`no file "${path}"; files: ${known.join(', ') || 'none'}`) + } + if (f.encoding === 'base64' && !isTextMime(f.mimeType)) { + return `"${f.path}" is a binary file (${f.mimeType}, ${f.size} bytes) and cannot be read as text.` + } + const text = f.encoding === 'base64' ? atob(f.content) : f.content + return text.length > maxRead + ? `${text.slice(0, maxRead)}… [truncated ${text.length - maxRead} chars]` + : text + }, + }), + ), + } + if (opts.readOnly) return tools + tools[`${p}write`] = tool({ + description: + 'Write a text file to the workspace (creates or replaces it). Use it for reports, drafts, data the user should be able to open.', + inputSchema: jsonSchema<{ path: string; content: string; mimeType?: string }>({ + type: 'object', + properties: { + path: { type: 'string', description: 'Absolute path, e.g. "/reports/summary.md"' }, + content: { type: 'string', description: 'The full file content' }, + mimeType: { type: 'string', description: 'Optional MIME type (guessed from the name)' }, + }, + required: ['path', 'content'], + additionalProperties: false, + }), + execute: async ({ path, content, mimeType }) => { + const info = await vfs.write(path, content, mimeType ? { mimeType } : {}) + return { written: info.path, size: info.size } + }, + }) + tools[`${p}delete`] = tool({ + description: 'Delete a file from the workspace.', + inputSchema: jsonSchema<{ path: string }>({ + type: 'object', + properties: { path: { type: 'string' } }, + required: ['path'], + additionalProperties: false, + }), + execute: async ({ path }) => ({ deleted: await vfs.delete(path) }), + }) + return tools +} diff --git a/src/files/vfs.ts b/src/files/vfs.ts new file mode 100644 index 0000000..b531526 --- /dev/null +++ b/src/files/vfs.ts @@ -0,0 +1,226 @@ +import { FILES_STORE, openAgentWebDB } from '../storage/db.js' + +/** + * A small virtual file system for the browser — the agent's (and the user's) + * workspace: attachments the user drops in, reports the agent writes, images + * pasted into the chat. Persistent in IndexedDB by default, in-memory on + * request (tests, private mode). Paths are POSIX-like and always absolute + * ("/notes/todo.md"); there are no real directories — a "directory" is a path + * prefix. + */ +export interface VirtualFile { + path: string + /** Text, or base64 for binary content (see `encoding`). */ + content: string + encoding: 'utf8' | 'base64' + mimeType: string + /** Size of the decoded content in bytes. */ + size: number + updatedAt: number +} + +/** A listing entry (no content). */ +export type VirtualFileInfo = Omit + +export interface VirtualFileSystemOptions { + /** Keep files in memory only (default false: IndexedDB). */ + memory?: boolean + /** IndexedDB database name, forwarded to the shared owner (default 'agent-web'). */ + dbName?: string + /** Isolates several file systems in one database (default 'default'). */ + namespace?: string + /** Refuse a single file larger than this many bytes (default 10 MB). */ + maxFileBytes?: number +} + +export type VfsListener = (change: { type: 'write' | 'delete'; path: string }) => void + +const MIME: Record = { + txt: 'text/plain', + md: 'text/markdown', + json: 'application/json', + csv: 'text/csv', + html: 'text/html', + css: 'text/css', + js: 'text/javascript', + ts: 'text/typescript', + svg: 'image/svg+xml', + png: 'image/png', + jpg: 'image/jpeg', + jpeg: 'image/jpeg', + gif: 'image/gif', + webp: 'image/webp', + pdf: 'application/pdf', + xml: 'application/xml', + yaml: 'text/yaml', + yml: 'text/yaml', +} + +/** A best-effort MIME type from a file name. */ +export const mimeFromPath = (path: string): string => + MIME[path.split('.').pop()?.toLowerCase() ?? ''] ?? 'application/octet-stream' + +/** True for MIME types the agent can read as text. */ +export const isTextMime = (mime: string): boolean => + /^text\/|json|xml|yaml|javascript|typescript|svg/.test(mime) + +/** + * Normalise a path: absolute, forward slashes, no "." / ".." / empty segments. + * Throws on a path that escapes the root. + */ +export const normalizePath = (input: string): string => { + const parts: string[] = [] + for (const seg of String(input ?? '') + .replace(/\\/g, '/') + .split('/')) { + if (!seg || seg === '.') continue + if (seg === '..') { + if (parts.length === 0) throw new Error(`path "${input}" escapes the root`) + parts.pop() + continue + } + parts.push(seg) + } + if (parts.length === 0) throw new Error('a file path is required') + return `/${parts.join('/')}` +} + +const byteLength = (content: string, encoding: VirtualFile['encoding']): number => + encoding === 'base64' + ? Math.floor((content.replace(/=+$/, '').length * 3) / 4) + : new TextEncoder().encode(content).length + +/** Base64 of bytes, chunked so a large file can't blow the call stack. */ +export const bytesToBase64 = (bytes: Uint8Array): string => { + let bin = '' + for (let i = 0; i < bytes.length; i += 0x8000) { + bin += String.fromCharCode(...bytes.subarray(i, i + 0x8000)) + } + return btoa(bin) +} + +export class VirtualFileSystem { + private readonly memory?: Map + private readonly dbName?: string + private readonly ns: string + private readonly maxFileBytes: number + private readonly listeners = new Set() + + constructor(opts: VirtualFileSystemOptions = {}) { + if (opts.memory) this.memory = new Map() + this.dbName = opts.dbName + this.ns = opts.namespace ?? 'default' + this.maxFileBytes = opts.maxFileBytes ?? 10 * 1024 * 1024 + } + + private key(path: string): string { + return `${this.ns}:${path}` + } + + /** Subscribe to writes and deletes (e.g. to refresh a file list). Returns an unsubscribe. */ + onChange(listener: VfsListener): () => void { + this.listeners.add(listener) + return () => this.listeners.delete(listener) + } + + private notify(type: 'write' | 'delete', path: string): void { + for (const l of this.listeners) { + try { + l({ type, path }) + } catch { + /* a bad listener must not break a write */ + } + } + } + + async write( + path: string, + content: string | Uint8Array, + opts: { mimeType?: string; encoding?: VirtualFile['encoding'] } = {}, + ): Promise { + const p = normalizePath(path) + const binary = content instanceof Uint8Array + const encoding = binary ? 'base64' : (opts.encoding ?? 'utf8') + const text = binary ? bytesToBase64(content) : content + const size = binary ? content.length : byteLength(text, encoding) + if (size > this.maxFileBytes) { + throw new Error(`"${p}" is ${size} bytes; the limit is ${this.maxFileBytes}`) + } + const file: VirtualFile = { + path: p, + content: text, + encoding, + mimeType: opts.mimeType ?? mimeFromPath(p), + size, + updatedAt: Date.now(), + } + if (this.memory) this.memory.set(p, file) + else { + const db = await openAgentWebDB({ dbName: this.dbName }) + await db.put(FILES_STORE, file, this.key(p)) + } + this.notify('write', p) + const { content: _drop, ...info } = file + return info + } + + /** Store a data URL ("data:image/png;base64,…") as a file. */ + async writeDataUrl(path: string, dataUrl: string): Promise { + const m = /^data:([^;,]+)?(;base64)?,(.*)$/s.exec(dataUrl) + if (!m) throw new Error('not a data URL') + const mimeType = m[1] ?? 'application/octet-stream' + return m[2] + ? this.write(path, m[3], { mimeType, encoding: 'base64' }) + : this.write(path, decodeURIComponent(m[3]), { mimeType }) + } + + async read(path: string): Promise { + const p = normalizePath(path) + if (this.memory) return this.memory.get(p) + const db = await openAgentWebDB({ dbName: this.dbName }) + return (await db.get(FILES_STORE, this.key(p))) as VirtualFile | undefined + } + + /** The file as a data URL (for , downloads). */ + async readDataUrl(path: string): Promise { + const f = await this.read(path) + if (!f) return undefined + return f.encoding === 'base64' + ? `data:${f.mimeType};base64,${f.content}` + : `data:${f.mimeType};charset=utf-8,${encodeURIComponent(f.content)}` + } + + async delete(path: string): Promise { + const p = normalizePath(path) + let existed: boolean + if (this.memory) existed = this.memory.delete(p) + else { + const db = await openAgentWebDB({ dbName: this.dbName }) + existed = (await db.get(FILES_STORE, this.key(p))) !== undefined + await db.delete(FILES_STORE, this.key(p)) + } + if (existed) this.notify('delete', p) + return existed + } + + /** Files under a prefix ("/" = all), sorted by path. */ + async list(prefix = '/'): Promise { + const dir = prefix === '/' ? '/' : `${normalizePath(prefix)}/` + let files: VirtualFile[] + if (this.memory) files = [...this.memory.values()] + else { + const db = await openAgentWebDB({ dbName: this.dbName }) + const range = IDBKeyRange.bound(`${this.ns}:`, `${this.ns}:￿`) + files = (await db.getAll(FILES_STORE, range)) as VirtualFile[] + } + return files + .filter((f) => dir === '/' || f.path.startsWith(dir)) + .map(({ content: _drop, ...info }) => info) + .sort((a, b) => a.path.localeCompare(b.path)) + } + + /** Remove every file (of this namespace). */ + async clear(): Promise { + for (const f of await this.list()) await this.delete(f.path) + } +} diff --git a/src/images.ts b/src/images.ts new file mode 100644 index 0000000..a67a5b1 --- /dev/null +++ b/src/images.ts @@ -0,0 +1,142 @@ +import type { FilePart } from 'ai' + +/** + * Attachments sent with a run: images, PDFs and other files the provider can + * read, given as bytes / base64 / a data URL — or as an http(s) URL, which + * providers that accept links (Gemini, Claude, OpenAI) fetch themselves and the + * AI SDK downloads for the rest (from the browser: subject to CORS). + */ +export interface RunFile { + /** Base64 (no prefix), a data URL ("data:image/png;base64,…"), raw bytes, or an http(s) URL. */ + data: string | Uint8Array | URL + /** e.g. "image/png", "application/pdf" — read from a data URL when omitted. */ + mediaType?: string + /** Shown in the transcript, in memory and in errors. */ + name?: string +} + +/** An image attachment (same shape; kept as its own name for clarity). */ +export type RunImage = RunFile + +/** The kinds of attachment the agent reasons about separately. */ +export type AttachmentKind = 'image' | 'pdf' | 'file' + +const DATA_URL_RE = /^data:([^;,]+)?;base64,(.*)$/s +const HTTP_RE = /^https?:\/\//i + +/** The media type of an attachment (declared, from a data URL, or from a URL's extension). */ +export const mediaTypeOf = (f: RunFile): string | undefined => { + if (f.mediaType) return f.mediaType + if (typeof f.data === 'string') { + const m = DATA_URL_RE.exec(f.data) + if (m?.[1]) return m[1] + } + const url = f.data instanceof URL ? f.data.href : typeof f.data === 'string' ? f.data : '' + if (HTTP_RE.test(url) || f.name) { + const ext = (f.name ?? new URL(url, 'http://x').pathname).split('.').pop()?.toLowerCase() + if (ext === 'pdf') return 'application/pdf' + if (ext && ['png', 'jpg', 'jpeg', 'gif', 'webp'].includes(ext)) { + return `image/${ext === 'jpg' ? 'jpeg' : ext}` + } + } + return undefined +} + +export const attachmentKind = (f: RunFile): AttachmentKind => { + const mt = mediaTypeOf(f) ?? '' + if (mt.startsWith('image/')) return 'image' + if (mt === 'application/pdf') return 'pdf' + return 'file' +} + +const KIND_LABEL: Record = { + image: 'images', + pdf: 'PDF files', + file: 'file attachments', +} + +const KIND_HINT: Record = { + image: + 'Remove the image, or switch to a vision-capable model such as Gemini, Claude or GPT-4o/GPT-5.', + pdf: 'Convert the PDF to text first (e.g. PDF → Markdown), or switch to a model that reads PDFs such as Gemini, Claude or GPT-4o/GPT-5.', + file: 'Paste its text instead, or switch to a model that accepts this file type.', +} + +/** Raised when attachments are sent to a model that cannot take them. */ +export class AttachmentsNotSupportedError extends Error { + readonly kind: AttachmentKind + constructor(model: string, kind: AttachmentKind, detail?: string) { + super( + `The model "${model}" can't take ${KIND_LABEL[kind]}${detail ? ` (${detail})` : ''}. ${KIND_HINT[kind]}`, + ) + this.name = 'AttachmentsNotSupportedError' + this.kind = kind + } + + toJSON(): string { + return this.message + } +} + +/** The image flavour of {@link AttachmentsNotSupportedError}. */ +export class ImagesNotSupportedError extends AttachmentsNotSupportedError { + constructor(model: string, detail?: string) { + super(model, 'image', detail) + this.name = 'ImagesNotSupportedError' + } +} + +/** The error for a kind: images get the dedicated class. */ +export const unsupportedAttachment = ( + model: string, + kind: AttachmentKind, + detail?: string, +): AttachmentsNotSupportedError => + kind === 'image' + ? new ImagesNotSupportedError(model, detail) + : new AttachmentsNotSupportedError(model, kind, detail) + +const dataOf = (f: RunFile): string | Uint8Array | URL => { + if (typeof f.data === 'string') { + const m = DATA_URL_RE.exec(f.data) + if (m) return m[2] + if (HTTP_RE.test(f.data)) return new URL(f.data) + } + return f.data +} + +/** + * An AI SDK file part (images included — the v7 SDK deprecates the separate + * image part). An http(s) URL is passed as a URL: providers that accept links + * fetch it themselves, and the SDK downloads it for the rest. + */ +export const toFilePart = (f: RunFile): FilePart => { + const kind = attachmentKind(f) + const mediaType = mediaTypeOf(f) ?? (kind === 'image' ? 'image' : 'application/octet-stream') + return { + type: 'file', + data: dataOf(f), + mediaType, + ...(f.name ? { filename: f.name } : {}), + } +} + +/** Back-compat name. */ +export const toImagePart = toFilePart + +/** A short, human label of a model for messages. */ +export const modelLabel = (model: unknown): string => { + if (typeof model === 'string') return model + const m = model as { provider?: string; modelId?: string } | undefined + return m?.modelId ? `${m.modelId}${m.provider ? ` (${m.provider})` : ''}` : 'this model' +} + +const REFUSAL_RE = + /image|vision|multimodal|multi-modal|media type|mediatype|pdf|document|unsupported (file|content|part|mime)|does not support (file|content)|invalid content type|image_url|inline_?data|file_?data/i + +/** Does a provider error look like a refusal of attachment input? */ +export const isAttachmentRefusal = (err: unknown): boolean => + REFUSAL_RE.test(err instanceof Error ? `${err.name} ${err.message}` : String(err)) + +/** Back-compat name. */ +export const isImageRefusal = isAttachmentRefusal diff --git a/src/index.ts b/src/index.ts index 8c5512c..f6f58a0 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,6 +1,6 @@ // ── Agent ────────────────────────────────────────────────────────────────── export { createAgent } from './agent/runner.js' -export type { Agent, RunOptions, RunResult } from './agent/runner.js' +export type { Agent, CompactOptions, CompactResult, RunOptions, RunResult } from './agent/runner.js' export type { IPlan, IPlanStep, IStepResult, IToolCall, IUsage } from './agent/loop-types.js' export { PlanSchema, PlanStepSchema, ReplanSchema } from './agent/schemas.js' export type { PlanShape, ReplanShape } from './agent/schemas.js' @@ -9,7 +9,75 @@ export { createPlan } from './agent/planner.js' export { executeStep, replanWanted, shouldReplan, splitBlocker } from './agent/executor.js' export { decideReplan } from './agent/replanner.js' export { synthesizeAnswer } from './agent/synthesizer.js' -export type { AgentContext } from './agent/internal.js' +export type { AgentContext, EffectiveToolStrategy } from './agent/internal.js' + +// ── Subagents (in-process or Web Worker) ───────────────────────────────────── +export { createSubagentTool } from './subagent/tool.js' +export type { SubagentToolOptions, SubagentWorkerConfig } from './subagent/tool.js' +export { serveSubagentWorker } from './subagent/worker.js' +export type { ServeSubagentOptions } from './subagent/worker.js' +export type { + MessageEndpoint, + WorkerLike, + ParentToWorker, + WorkerToParent, + ProxiedToolSpec, +} from './subagent/protocol.js' + +// ── Images, files ──────────────────────────────────────────────────────────── +export { + AttachmentsNotSupportedError, + ImagesNotSupportedError, + attachmentKind, + isAttachmentRefusal, + isImageRefusal, + mediaTypeOf, + toFilePart, + toImagePart, +} from './images.js' +export type { AttachmentKind, RunFile, RunImage } from './images.js' +export { + VirtualFileSystem, + normalizePath, + mimeFromPath, + isTextMime, + bytesToBase64, +} from './files/vfs.js' +export type { + VirtualFile, + VirtualFileInfo, + VirtualFileSystemOptions, + VfsListener, +} from './files/vfs.js' +export { createFileTools } from './files/tools.js' +export type { FileToolsOptions } from './files/tools.js' + +// ── Thinking, limits, caching ──────────────────────────────────────────────── +export { resolveThinking, thinkingFor, mergeProviderOptions } from './thinking.js' +export type { + ThinkingConfig, + ThinkingLevel, + ThinkingSetting, + ThinkingStage, + ResolvedThinking, + ProviderOptionsMap, +} from './thinking.js' +export { checkLimits, ToolBudgetError } from './limits.js' +export type { TokenLimits, LimitBreach, LimitKind } from './limits.js' +export { resolveCaching, withRollingBreakpoint } from './caching.js' +export type { PromptCachingSetting, ResolvedCaching } from './caching.js' + +// ── Skills ─────────────────────────────────────────────────────────────────── +export { + defineSkill, + parseSkillMarkdown, + loadSkillFromUrl, + renderSkillIndex, + renderActiveSkills, + createSkillTools, + SKILL_TOOL_NAMES, +} from './skills.js' +export type { Skill, LoadSkillOptions } from './skills.js' // ── Config ─────────────────────────────────────────────────────────────────── export { resolveConfig } from './config.js' @@ -35,6 +103,8 @@ export { supportsNativeTools, supportsStructuredOutput, directBrowserOk, + supportsImages, + supportsPdf, } from './providers/capabilities.js' export { isDirectModel, isProviderSpec } from './providers/types.js' export type { @@ -55,7 +125,13 @@ export { getOrCreateVaultKey, encryptJSON, decryptJSON } from './secrets/crypto. export type { EncryptedBlob } from './secrets/crypto.js' // ── Storage (shared IndexedDB owner) ───────────────────────────────────────── -export { openAgentWebDB, KEYS_STORE, SECRETS_STORE, SESSIONS_STORE } from './storage/db.js' +export { + openAgentWebDB, + KEYS_STORE, + SECRETS_STORE, + SESSIONS_STORE, + FILES_STORE, +} from './storage/db.js' export type { AgentWebDBOptions } from './storage/db.js' // ── Low-level LLM helpers (simple generation, tool loop) ───────────────────── @@ -71,14 +147,49 @@ export { renderCatalog, dispatch } from './tools/prompted.js' export { selectToolMode } from './tools/mode.js' export { promptHintOf } from './tools/types.js' export type { AgentTool, AgentToolSet } from './tools/types.js' +export { + ToolDeniedError, + markReadOnly, + isReadOnlyTool, + matchRule, + decideToolPermission, + TOOL_APPROVAL_MODES, +} from './tools/approval.js' +export type { + ToolApprovalConfig, + ToolApprovalDecision, + ToolApprovalMode, + ToolApprovalRequest, + ToolPermission, +} from './tools/approval.js' +export { + searchTools, + renderSearchCatalog, + toolCatalogOf, + createFindToolsTool, + tokenize, + FIND_TOOLS_NAME, +} from './tools/search.js' +export type { ToolCatalogEntry, SearchToolsOptions } from './tools/search.js' +export { toolRunContextOf, schemaHintOf, limitModelOutput } from './tools/wrap.js' +export { createToolResultClearer } from './llm/context-editing.js' +export type { ToolResultClearing, ClearedInfo } from './llm/context-editing.js' +export type { ToolRunContext } from './tools/wrap.js' // ── Memory ─────────────────────────────────────────────────────────────────── export { MemoryStore } from './memory/store.js' export type { ContextStore, StoredMessage } from './memory/store.js' export { IndexedDBStore } from './memory/sessions.js' export type { IndexedDBStoreOptions } from './memory/sessions.js' -export { compressHistory } from './memory/compress.js' -export type { CompressOptions } from './memory/compress.js' +export { + compressHistory, + compactHistory, + compactSteps, + estimateTokens, + estimateMessagesTokens, + resolveCompaction, +} from './memory/compress.js' +export type { CompressOptions, CompactionConfig, ResolvedCompaction } from './memory/compress.js' // ── Prompts ────────────────────────────────────────────────────────────────── export { defaultPrompts, withSystem } from './prompts.js' @@ -110,7 +221,7 @@ export type { } from './parse.js' // ── Events ─────────────────────────────────────────────────────────────────── -export type { AgentEvent, AgentEventHandler, ReplanMode, Phase } from './events.js' +export type { AgentEvent, AgentEventHandler, ReplanMode, Phase, UsagePhase } from './events.js' // ── Logging ────────────────────────────────────────────────────────────────── export { createLogger } from './logger.js' diff --git a/src/limits.ts b/src/limits.ts new file mode 100644 index 0000000..30c914b --- /dev/null +++ b/src/limits.ts @@ -0,0 +1,79 @@ +import type { IUsage } from './agent/loop-types.js' + +/** + * Run-level token budgets. Every cap is cumulative over one `run()` and soft: + * it is checked between steps and at every tool-calling round inside a step, + * so a run stops at the next boundary after crossing it and then writes its + * final answer (synthesis still runs, bounded by `perCall.synthesizer`). + */ +export interface TokenLimits { + /** Cumulative input (prompt) tokens per run. */ + maxInputTokens?: number + /** Cumulative output tokens per run, thinking included. */ + maxOutputTokens?: number + /** Cumulative thinking tokens per run. */ + maxReasoningTokens?: number + /** Cumulative input + output tokens per run. */ + maxTotalTokens?: number + /** Max output tokens of ONE model call, per phase (wins over the legacy `budgets`). */ + perCall?: { + planner?: number + executor?: number + replanner?: number + synthesizer?: number + compaction?: number + } +} + +export type LimitKind = 'input' | 'output' | 'reasoning' | 'total' | 'tool-calls' + +export interface LimitBreach { + kind: LimitKind + tokens: number + cap: number +} + +const over = (value: number | undefined, cap: number | undefined): boolean => + typeof cap === 'number' && cap > 0 && (value ?? 0) >= cap + +/** + * The first cap the usage has reached, or undefined. Order: total, input, + * output, reasoning — the broadest budget is reported first. Pure. + */ +export const checkLimits = ( + usage: IUsage, + limits: TokenLimits | undefined, +): LimitBreach | undefined => { + if (!limits) return undefined + if (over(usage.totalTokens, limits.maxTotalTokens)) { + return { kind: 'total', tokens: usage.totalTokens, cap: limits.maxTotalTokens as number } + } + if (over(usage.inputTokens, limits.maxInputTokens)) { + return { kind: 'input', tokens: usage.inputTokens, cap: limits.maxInputTokens as number } + } + if (over(usage.outputTokens, limits.maxOutputTokens)) { + return { kind: 'output', tokens: usage.outputTokens, cap: limits.maxOutputTokens as number } + } + if (over(usage.reasoningTokens, limits.maxReasoningTokens)) { + return { + kind: 'reasoning', + tokens: usage.reasoningTokens ?? 0, + cap: limits.maxReasoningTokens as number, + } + } + return undefined +} + +/** Raised by a tool call that would exceed `maxToolCalls`. */ +export class ToolBudgetError extends Error { + constructor(cap: number) { + super( + `Tool-call budget exhausted (${cap} per run). Do not call more tools; report what you have.`, + ) + this.name = 'ToolBudgetError' + } + + toJSON(): string { + return this.message + } +} diff --git a/src/llm/context-editing.ts b/src/llm/context-editing.ts new file mode 100644 index 0000000..a5bea9c --- /dev/null +++ b/src/llm/context-editing.ts @@ -0,0 +1,94 @@ +import type { ModelMessage } from 'ai' +import { estimateTokens } from '../memory/compress.js' + +/** + * Context editing inside one tool loop — what Anthropic's + * `clear_tool_uses` and Claude Code's micro-compaction do: once the loop's + * conversation grows past `triggerTokens`, the OLDEST tool results are replaced + * with a one-line stub (the calls themselves stay, so the model knows what it + * did and can call again), keeping the `keep` most recent results verbatim. + * + * Clearing is sticky — a result cleared once stays cleared — so the prefix the + * provider cached does not flip back and forth between rounds. + */ +export interface ToolResultClearing { + /** Clear once the loop's estimated tokens exceed this. */ + triggerTokens: number + /** Most recent tool results kept verbatim (default 3). */ + keep?: number +} + +export interface ClearedInfo { + cleared: number + beforeTokens: number + afterTokens: number +} + +const STUB = (name: string): string => + `[${name} result cleared to save context — call the tool again if you still need it]` + +const estimate = (messages: ModelMessage[]): number => { + try { + return estimateTokens(JSON.stringify(messages)) + } catch { + return 0 + } +} + +// A result's position in the conversation ("message:part"). The loop only +// appends, so positions are stable — unlike tool call ids, which some servers +// repeat across rounds. +const positionsOf = (messages: ModelMessage[]): string[] => + messages.flatMap((m, i) => + m.role === 'tool' && Array.isArray(m.content) + ? (m.content as { type: string }[]).flatMap((p, j) => + p.type === 'tool-result' ? [`${i}:${j}`] : [], + ) + : [], + ) + +/** + * A stateful clearer for one loop: call it with each round's messages; it + * returns the (possibly) edited copy and reports what it newly cleared. + */ +export const createToolResultClearer = ( + opts: ToolResultClearing, + onCleared?: (info: ClearedInfo) => void, +): ((messages: ModelMessage[]) => ModelMessage[]) => { + const keep = Math.max(0, opts.keep ?? 3) + const cleared = new Set() + + const apply = (messages: ModelMessage[]): ModelMessage[] => + cleared.size === 0 + ? messages + : messages.map((m, i) => + m.role === 'tool' && Array.isArray(m.content) + ? { + ...m, + content: m.content.map((p, j) => + p.type === 'tool-result' && cleared.has(`${i}:${j}`) + ? { ...p, output: { type: 'text' as const, value: STUB(p.toolName) } } + : p, + ), + } + : m, + ) + + return (messages) => { + let edited = apply(messages) + const before = estimate(edited) + if (before <= opts.triggerTokens) return edited + const positions = positionsOf(messages) + let added = 0 + for (const pos of positions.slice(0, Math.max(0, positions.length - keep))) { + if (!cleared.has(pos)) { + cleared.add(pos) + added += 1 + } + } + if (added === 0) return edited + edited = apply(messages) + onCleared?.({ cleared: added, beforeTokens: before, afterTokens: estimate(edited) }) + return edited + } +} diff --git a/src/llm/generate.ts b/src/llm/generate.ts index 84e85e5..f3260b2 100644 --- a/src/llm/generate.ts +++ b/src/llm/generate.ts @@ -4,24 +4,41 @@ import { streamText, type LanguageModel, type ModelMessage, + type SystemModelMessage, type ToolSet, } from 'ai' import type { ZodType } from 'zod' +import type { ProviderOptionsMap, ThinkingLevel } from '../thinking.js' import { promptOf, timeoutSignal } from './util.js' /** Options shared by the low-level generation helpers. Provide `prompt` OR `messages`. */ export interface GenerateOptions { - system?: string + /** The system prompt — a string, or a SystemModelMessage (e.g. with a cache breakpoint). */ + system?: string | SystemModelMessage prompt?: string messages?: ModelMessage[] tools?: ToolSet maxOutputTokens?: number temperature?: number + /** Portable thinking level (see `resolveThinking`). */ + reasoning?: ThinkingLevel + /** Provider-keyed options (thinking budgets, cache keys, …). */ + providerOptions?: ProviderOptionsMap abortSignal?: AbortSignal /** Time-box the call; combined with `abortSignal` (0/undefined = no timeout). */ timeoutMs?: number } +/** The settings every helper forwards; omits undefined keys the SDK would reject. */ +const callSettings = (opts: GenerateOptions) => ({ + ...(opts.system !== undefined ? { instructions: opts.system } : {}), + maxOutputTokens: opts.maxOutputTokens, + temperature: opts.temperature, + ...(opts.reasoning ? { reasoning: opts.reasoning } : {}), + ...(opts.providerOptions ? { providerOptions: opts.providerOptions as never } : {}), + abortSignal: timeoutSignal(opts.abortSignal, opts.timeoutMs), +}) + /** * One-shot text generation. Thin, provider-agnostic wrapper over the AI SDK's * `generateText` — the same call works against a cloud model, a local WebLLM @@ -34,12 +51,9 @@ export const generate = ( ): ReturnType => generateText({ model, - system: opts.system, + ...callSettings(opts), ...promptOf(opts), tools: opts.tools, - maxOutputTokens: opts.maxOutputTokens, - temperature: opts.temperature, - abortSignal: timeoutSignal(opts.abortSignal, opts.timeoutMs), }) /** Streaming text generation. Returns the AI SDK `streamText` result (`.textStream`, `.fullStream`). */ @@ -49,12 +63,9 @@ export const stream = ( ): ReturnType => streamText({ model, - system: opts.system, + ...callSettings(opts), ...promptOf(opts), tools: opts.tools, - maxOutputTokens: opts.maxOutputTokens, - temperature: opts.temperature, - abortSignal: timeoutSignal(opts.abortSignal, opts.timeoutMs), }) /** @@ -71,9 +82,6 @@ export const generateStructured = ( generateObject({ model, schema, - system: opts.system, + ...callSettings(opts), ...promptOf(opts), - maxOutputTokens: opts.maxOutputTokens, - temperature: opts.temperature, - abortSignal: timeoutSignal(opts.abortSignal, opts.timeoutMs), }) diff --git a/src/llm/tool-loop.ts b/src/llm/tool-loop.ts index fb0e9ff..7151898 100644 --- a/src/llm/tool-loop.ts +++ b/src/llm/tool-loop.ts @@ -4,38 +4,72 @@ import { streamText, type LanguageModel, type ModelMessage, + type StopCondition, + type SystemModelMessage, type ToolSet, } from 'ai' import { addUsage, BLOCKER, emptyUsage, type IToolCall, type IUsage } from '../agent/loop-types.js' +import { withRollingBreakpoint, type ResolvedCaching } from '../caching.js' import { parseExecutorResponse } from '../parse.js' +import type { ProviderOptionsMap, ThinkingLevel } from '../thinking.js' import { dispatch } from '../tools/prompted.js' +import { + createToolResultClearer, + type ClearedInfo, + type ToolResultClearing, +} from './context-editing.js' import { clip, normalizeUsage, promptOf, timeoutSignal } from './util.js' export interface ToolLoopCallbacks { onTextDelta?: (delta: string) => void + /** The model's streamed thoughts (native mode, thinking enabled). */ + onReasoningDelta?: (delta: string) => void onToolCall?: (name: string, input: unknown) => void onToolResult?: (name: string, output: unknown, ok: boolean) => void + /** Older tool results were replaced with stubs (see `clearToolResults`). */ + onToolResultsCleared?: (info: ClearedInfo) => void } export interface ToolLoopOptions { /** 'native' = SDK function-calling; 'prompted' = parse JSON out of plain text. */ mode: 'native' | 'prompted' - system?: string + system?: string | SystemModelMessage prompt?: string messages?: ModelMessage[] /** In native mode: the callable tools. In prompted mode: used to dispatch salvaged calls. */ tools: ToolSet /** - * Restrict callable tools to these names (plan-narrowed). Native mode passes - * them to the SDK's `activeTools`; prompted mode bounds `dispatch` to them. + * Restrict callable tools to these names (plan-narrowed / search). Native mode + * passes them to the SDK's `activeTools`; prompted mode bounds `dispatch` to + * them. A function is re-read before every round, so a tool that activates + * others (find_tools) takes effect on the next round. + */ + activeTools?: string[] | (() => string[]) + /** + * Prompted mode: render catalogue lines for tools that became active during + * the call, so the model learns their names and parameters. */ - activeTools?: string[] + describeTools?: (names: string[]) => string /** Cap on tool-calling rounds within this call, in both modes (default 4). */ maxSteps?: number + /** + * Checked with the usage of the rounds so far, at every round boundary — + * return true to stop (e.g. a run-level token budget is spent). + */ + shouldStop?: (usageSoFar: IUsage) => boolean maxOutputTokens?: number temperature?: number + reasoning?: ThinkingLevel + providerOptions?: ProviderOptionsMap abortSignal?: AbortSignal timeoutMs?: number + /** + * Prompt caching: with it enabled, every round moves a cache breakpoint to + * the newest message, so the rounds before it are read from the cache. + */ + caching?: ResolvedCaching + /** Clear the oldest tool results once the loop's context grows past a threshold. */ + clearToolResults?: ToolResultClearing callbacks?: ToolLoopCallbacks } @@ -43,6 +77,8 @@ export interface ToolLoopResult { text: string toolCalls: IToolCall[] usage: IUsage + /** True when `shouldStop` cut the call short. */ + stoppedByBudget?: boolean } /** @@ -59,19 +95,62 @@ export const runToolLoop = ( ): Promise => opts.mode === 'prompted' ? runPrompted(model, opts) : runNative(model, opts) +const activeOf = (opts: ToolLoopOptions): string[] | undefined => + typeof opts.activeTools === 'function' ? opts.activeTools() : opts.activeTools + +const callExtras = (opts: ToolLoopOptions) => ({ + ...(opts.system !== undefined ? { instructions: opts.system } : {}), + maxOutputTokens: opts.maxOutputTokens, + temperature: opts.temperature, + ...(opts.reasoning ? { reasoning: opts.reasoning } : {}), + ...(opts.providerOptions ? { providerOptions: opts.providerOptions as never } : {}), +}) + const runNative = async (model: LanguageModel, opts: ToolLoopOptions): Promise => { const toolCalls: IToolCall[] = [] const pending = new Map() + let stoppedByBudget = false + + const stopWhen: StopCondition[] = [stepCountIs(opts.maxSteps ?? 4)] + if (opts.shouldStop) { + const shouldStop = opts.shouldStop + stopWhen.push(({ steps }) => { + const used = steps.reduce((u, s) => addUsage(u, normalizeUsage(s.usage)), emptyUsage()) + if (!shouldStop(used)) return false + stoppedByBudget = true + return true + }) + } + const initialActive = activeOf(opts) + const dynamic = typeof opts.activeTools === 'function' + const caching = opts.caching?.enabled ? opts.caching : undefined + const clear = opts.clearToolResults + ? createToolResultClearer(opts.clearToolResults, opts.callbacks?.onToolResultsCleared) + : undefined + const editMessages = caching || clear const result = streamText({ model, tools: opts.tools, - ...(opts.activeTools ? { activeTools: opts.activeTools } : {}), - stopWhen: stepCountIs(opts.maxSteps ?? 4), - system: opts.system, + ...(initialActive ? { activeTools: initialActive } : {}), + // Before every round: re-read the active set so tools activated mid-call + // (find_tools) become callable; clear stale tool results; move the cache + // breakpoint to the newest message. + ...(dynamic || editMessages + ? { + prepareStep: ({ messages }: { messages: ModelMessage[] }) => { + let next = clear ? clear(messages) : messages + if (caching) next = withRollingBreakpoint(next, caching) + return { + ...(dynamic ? { activeTools: activeOf(opts) } : {}), + ...(editMessages ? { messages: next } : {}), + } + }, + } + : {}), + stopWhen, + ...callExtras(opts), ...promptOf(opts), - maxOutputTokens: opts.maxOutputTokens, - temperature: opts.temperature, abortSignal: timeoutSignal(opts.abortSignal, opts.timeoutMs), }) @@ -80,6 +159,9 @@ const runNative = async (model: LanguageModel, opts: ToolLoopOptions): Promise { } /** One round's results + the continue-or-finish contract for the next round. */ -const toolResultsPrompt = (results: IToolCall[]): string => { +const toolResultsPrompt = (results: IToolCall[], newTools: string): string => { const lines = results.map( (r) => `- ${r.name} ${clip(asJson(r.input), 160)} → ${r.ok ? 'ok' : 'FAILED'}: ${clip(asJson(r.output), 400)}`, ) return `TOOL RESULTS: ${lines.join('\n')} - +${newTools ? `\nTOOLS NOW AVAILABLE (call them by these exact names):\n${newTools}\n` : ''} Continue THIS step using the results above. Reply with a single JSON object, nothing else: { "reply": string, "actions": [ { "tool": string, "args": object } ] } - If the step is now complete: "actions": [] and a short outcome in "reply". @@ -156,10 +238,12 @@ const runPrompted = async ( // never learns what its calls returned — a read-tool would be write-only. Set // maxSteps: 1 for the old single-round behaviour. const maxRounds = Math.max(1, opts.maxSteps ?? 4) - const active = opts.activeTools - const tools = active - ? Object.fromEntries(Object.entries(opts.tools).filter(([name]) => active.includes(name))) - : opts.tools + const callable = (): ToolSet => { + const active = activeOf(opts) + return active + ? Object.fromEntries(Object.entries(opts.tools).filter(([name]) => active.includes(name))) + : opts.tools + } // One watchdog for the WHOLE call (as native mode does with its single // stream), so a slow local model can't run up to maxRounds × timeoutMs; every @@ -177,15 +261,19 @@ const runPrompted = async ( const seen = new Set() let usage = emptyUsage() let text = '' + let stoppedByBudget = false for (let round = 1; round <= maxRounds; round += 1) { if (signal?.aborted) break + if (opts.shouldStop?.(usage)) { + stoppedByBudget = true + break + } + const before = new Set(Object.keys(callable())) const result = await generateText({ model, - system: opts.system, - messages, - maxOutputTokens: opts.maxOutputTokens, - temperature: opts.temperature, + ...callExtras(opts), + messages: opts.caching ? withRollingBreakpoint(messages, opts.caching) : messages, abortSignal: signal, }) usage = addUsage(usage, normalizeUsage(result.usage)) @@ -212,15 +300,24 @@ const runPrompted = async ( for (const action of parsed.actions) { if (signal?.aborted) break // stop mid-batch the moment the caller aborts opts.callbacks?.onToolCall?.(action.tool, action.args) - const rec = await dispatch(action, tools, { abortSignal: signal }) + // Re-read per action: an earlier action of this batch may have + // activated more tools (find_tools). + const rec = await dispatch(action, callable(), { abortSignal: signal }) toolCalls.push(rec) results.push(rec) opts.callbacks?.onToolResult?.(rec.name, rec.output, rec.ok) } if (round === maxRounds || signal?.aborted) break + const added = Object.keys(callable()).filter((n) => !before.has(n)) messages.push({ role: 'assistant', content: result.text }) - messages.push({ role: 'user', content: toolResultsPrompt(results) }) + messages.push({ + role: 'user', + content: toolResultsPrompt( + results, + added.length && opts.describeTools ? opts.describeTools(added) : '', + ), + }) } // Prompted mode can't stream, so surface the final reply ONCE (not per round) @@ -228,5 +325,5 @@ const runPrompted = async ( // otherwise concatenate every round's full reply. if (text) opts.callbacks?.onTextDelta?.(text) - return { text, toolCalls, usage } + return { text, toolCalls, usage, stoppedByBudget } } diff --git a/src/llm/util.ts b/src/llm/util.ts index b038363..39a326f 100644 --- a/src/llm/util.ts +++ b/src/llm/util.ts @@ -15,18 +15,29 @@ export const promptOf = (opts: { }): { prompt: string } | { messages: ModelMessage[] } => opts.messages ? { messages: opts.messages } : { prompt: opts.prompt ?? '' } -/** Normalise the AI SDK usage shape (fields can be undefined) into IUsage. */ -export const normalizeUsage = (u?: { +/** The AI SDK usage shape, every field optional (providers omit what they don't report). */ +export interface SdkUsageLike { inputTokens?: number outputTokens?: number totalTokens?: number -}): IUsage => { + inputTokenDetails?: { cacheReadTokens?: number; cacheWriteTokens?: number } + outputTokenDetails?: { reasoningTokens?: number } + /** Pre-v6 flat fields, still emitted by some providers. */ + reasoningTokens?: number + cachedInputTokens?: number +} + +/** Normalise the AI SDK usage shape (fields can be undefined) into IUsage. */ +export const normalizeUsage = (u?: SdkUsageLike): IUsage => { const inputTokens = u?.inputTokens ?? 0 const outputTokens = u?.outputTokens ?? 0 return { inputTokens, outputTokens, totalTokens: u?.totalTokens ?? inputTokens + outputTokens, + reasoningTokens: u?.outputTokenDetails?.reasoningTokens ?? u?.reasoningTokens ?? 0, + cachedInputTokens: u?.inputTokenDetails?.cacheReadTokens ?? u?.cachedInputTokens ?? 0, + cacheWriteTokens: u?.inputTokenDetails?.cacheWriteTokens ?? 0, } } diff --git a/src/mcp/http.ts b/src/mcp/http.ts index 93b4e7d..fdd8679 100644 --- a/src/mcp/http.ts +++ b/src/mcp/http.ts @@ -43,12 +43,16 @@ export interface McpHttpServerConfig { fetch?: FetchLike /** Overrides for the SSE reconnect backoff. Merged over the defaults below. */ reconnection?: Partial + /** Give up on this server after this many ms of connect + listing (overrides the option). */ + connectTimeoutMs?: number } export interface McpCatalogEntry { name: string description: string server: string + /** The server marked the tool read-only (`annotations.readOnlyHint`). */ + readOnly?: boolean } export interface McpServerResult { @@ -88,6 +92,47 @@ export interface ConnectMcpOptions { * connector deliberately does not refresh behind a running agent's back. */ onToolsChanged?: (server: string) => void + /** + * Per-server deadline for connect + tool listing (default 30 000 ms). A + * server that misses it is reported failed and closed; the others still + * mount — one hanging server no longer holds every other one hostage. + */ + connectTimeoutMs?: number +} + +// Hard stop for a server whose pagination never ends (cursor loops). +const MAX_LIST_PAGES = 100 + +/** + * List every tool of a server, following `nextCursor`. Servers with hundreds + * of tools paginate; reading only the first page silently drops the rest. + */ +const listAllTools = async (client: Client): Promise => { + const all: McpToolDescriptor[] = [] + let cursor: string | undefined + for (let page = 0; page < MAX_LIST_PAGES; page += 1) { + const res = await client.listTools(cursor ? { cursor } : undefined) + all.push(...(res.tools as McpToolDescriptor[])) + cursor = res.nextCursor + if (!cursor) break + } + return all +} + +class ConnectTimeoutError extends Error {} + +const withTimeout = (p: Promise, ms: number, label: string): Promise => { + if (!(ms > 0)) return p + let timer: ReturnType | undefined + return Promise.race([ + p.finally(() => clearTimeout(timer)), + new Promise((_, reject) => { + timer = setTimeout( + () => reject(new ConnectTimeoutError(`${label} timed out after ${ms} ms`)), + ms, + ) + }), + ]) } // Providers cap tool names at 64 chars (^[a-zA-Z0-9_-]{1,64}$). We enforce the @@ -145,6 +190,7 @@ interface McpToolDescriptor { name: string description?: string inputSchema: unknown + annotations?: { readOnlyHint?: boolean; destructiveHint?: boolean; title?: string } } interface ServerConnect { @@ -307,60 +353,37 @@ const openConnection = async ( log: NonNullable, ): Promise => { let client: Client | undefined + let timedOut = false + const timeoutMs = cfg.connectTimeoutMs ?? opts.connectTimeoutMs ?? 30_000 + const attempt = connectAndList(name, cfg, opts, log, (c) => { + client = c + // Created after the deadline already passed: nobody will ever use it. + if (timedOut) void c.close().catch(() => {}) + }) + // A handshake that completes after the deadline must not leave a live session behind. + attempt.then( + (r) => { + if (timedOut) void r.client?.close().catch(() => {}) + }, + () => {}, + ) try { - if (cfg.headers && cfg.getHeaders) { - throw new Error(`MCP server "${name}": specify either headers or getHeaders, not both`) - } - const headers = cfg.getHeaders ? await cfg.getHeaders() : cfg.headers - if (cfg.authProvider) { - if (headers && 'Authorization' in headers) { - log( - 'warn', - `[mcp] ${name}: an explicit Authorization header shadows the OAuth token from authProvider`, - ) - } - assertSecureOAuthUrl(cfg.url, name) - await refreshIfExpired(cfg.authProvider, cfg.url, cfg.fetch, log, name) - } - const transport = new StreamableHTTPClientTransport(new URL(cfg.url), { - requestInit: headers ? { headers } : undefined, - authProvider: cfg.authProvider, - fetch: cfg.fetch, - reconnectionOptions: { ...DEFAULT_RECONNECTION, ...cfg.reconnection }, - }) - // Without these, a dropped SSE stream or a transport-level protocol error - // is swallowed and the agent just stops getting results. - transport.onerror = (err) => log('warn', `[mcp] ${name}: transport error - ${err.message}`) - // info, not warn: an ordinary close() ends here too, and a routine - // teardown logged as a warning trains people to ignore warnings. - transport.onclose = () => log('info', `[mcp] ${name}: transport closed`) - - client = new Client({ - name: opts.clientName ?? 'agent-web', - version: opts.clientVersion ?? '0.0.0', - }) - await client.connect(transport) - - // Subscribe BEFORE the first list call: a server that mutates its tool set - // during init would otherwise lose the notification in the window between - // connect() and listTools(). - if (opts.onToolsChanged) { - const notify = opts.onToolsChanged - client.setNotificationHandler(ToolListChangedNotificationSchema, () => notify(name)) - } - - const listed = (await client.listTools()).tools as McpToolDescriptor[] - return { name, client, listed } + return await withTimeout(attempt, timeoutMs, 'connect') } catch (err) { + if (err instanceof ConnectTimeoutError) timedOut = true // The client may be live even though we ended up here (listTools() failing - // after a successful connect). Close it, or the SSE stream leaks for the - // lifetime of the tab. + // after a successful connect, or the deadline passing mid-listing). Close + // it, or the SSE stream leaks for the lifetime of the tab. if (client) await client.close().catch(() => {}) const needsAuthorization = err instanceof UnauthorizedError let message = err instanceof Error ? err.message : String(err) // Turn "TypeError: Failed to fetch" into the sentence the reader needs. // Only on the failure path, so the extra request costs nothing in normal use. - if (!needsAuthorization && isNetworkLevelFailure(err)) { + if ( + !needsAuthorization && + !(err instanceof ConnectTimeoutError) && + isNetworkLevelFailure(err) + ) { const hint = await diagnoseMcpCors(cfg.url, cfg.fetch).catch(() => undefined) if (hint) message = `${message} - ${hint}` } @@ -374,6 +397,59 @@ const openConnection = async ( } } +const connectAndList = async ( + name: string, + cfg: McpHttpServerConfig, + opts: ConnectMcpOptions, + log: NonNullable, + onClient: (client: Client) => void, +): Promise => { + if (cfg.headers && cfg.getHeaders) { + throw new Error(`MCP server "${name}": specify either headers or getHeaders, not both`) + } + const headers = cfg.getHeaders ? await cfg.getHeaders() : cfg.headers + if (cfg.authProvider) { + if (headers && 'Authorization' in headers) { + log( + 'warn', + `[mcp] ${name}: an explicit Authorization header shadows the OAuth token from authProvider`, + ) + } + assertSecureOAuthUrl(cfg.url, name) + await refreshIfExpired(cfg.authProvider, cfg.url, cfg.fetch, log, name) + } + const transport = new StreamableHTTPClientTransport(new URL(cfg.url), { + requestInit: headers ? { headers } : undefined, + authProvider: cfg.authProvider, + fetch: cfg.fetch, + reconnectionOptions: { ...DEFAULT_RECONNECTION, ...cfg.reconnection }, + }) + // Without these, a dropped SSE stream or a transport-level protocol error + // is swallowed and the agent just stops getting results. + transport.onerror = (err) => log('warn', `[mcp] ${name}: transport error - ${err.message}`) + // info, not warn: an ordinary close() ends here too, and a routine + // teardown logged as a warning trains people to ignore warnings. + transport.onclose = () => log('info', `[mcp] ${name}: transport closed`) + + const client = new Client({ + name: opts.clientName ?? 'agent-web', + version: opts.clientVersion ?? '0.0.0', + }) + onClient(client) + await client.connect(transport) + + // Subscribe BEFORE the first list call: a server that mutates its tool set + // during init would otherwise lose the notification in the window between + // connect() and listTools(). + if (opts.onToolsChanged) { + const notify = opts.onToolsChanged + client.setNotificationHandler(ToolListChangedNotificationSchema, () => notify(name)) + } + + const listed = await listAllTools(client) + return { name, client, listed } +} + /** * Connect to one or more HTTP MCP servers and return their tools + catalogue. * Servers connect concurrently; tools, catalogue, and results merge in the @@ -428,6 +504,7 @@ export const connectMcpHttp = async ( log('warn', `[mcp] ${name}: tool name "${prefixed}" already taken; mounted as "${key}"`) } const description = t.description ?? '' + const readOnly = t.annotations?.readOnlyHint === true tools[key] = dynamicTool({ description, inputSchema: jsonSchema(t.inputSchema as Parameters[0]), @@ -447,7 +524,9 @@ export const connectMcpHttp = async ( return flat }, }) - catalog.push({ name: key, description, server: name }) + // The agent's consent gate lets read-only tools through in 'ask-writes'. + if (readOnly) (tools[key] as { readOnly?: boolean }).readOnly = true + catalog.push({ name: key, description, server: name, ...(readOnly ? { readOnly } : {}) }) keys.push(key) mounted += 1 } @@ -481,8 +560,7 @@ export const connectMcpHttp = async ( refreshServer: async (name: string) => { const client = clients.get(name) if (!client) throw new Error(`MCP server "${name}" is not connected`) - const listed = (await client.listTools()).tools as McpToolDescriptor[] - mountTools(name, client, listed) + mountTools(name, client, await listAllTools(client)) }, close: async () => { await Promise.allSettled([...clients.values()].map((c) => c.close())) diff --git a/src/memory/compress.ts b/src/memory/compress.ts index 3408f3f..6bede65 100644 --- a/src/memory/compress.ts +++ b/src/memory/compress.ts @@ -1,18 +1,96 @@ import type { LanguageModel } from 'ai' import { generate } from '../llm/generate.js' +import { normalizeUsage } from '../llm/util.js' +import type { IUsage } from '../agent/loop-types.js' import type { StoredMessage } from './store.js' +/** + * Context compaction. A model's window is finite and every token in it is paid + * for on every call, so long transcripts and long runs are folded into a + * summary: the oldest material is summarised by a model, the most recent is + * kept verbatim. + */ +export interface CompactionConfig { + /** Compact automatically before planning and between steps (default true). */ + auto?: boolean + /** The model's context window, in tokens (default 128 000). */ + contextWindowTokens?: number + /** Compact once the estimated context exceeds this (default: half the window). */ + thresholdTokens?: number + /** Transcript messages kept verbatim (default 4). */ + keepRecentTurns?: number + /** Executed steps kept verbatim in the run's trace (default 3). */ + keepRecentSteps?: number + /** Output cap of a summary call (default 1024). */ + summaryMaxTokens?: number + /** Cap on ONE tool result as the model sees it, in chars (default 20 000; 0 = no cap). */ + maxToolOutputChars?: number + /** + * Inside one step's tool loop: once its context passes this many tokens, + * replace the oldest tool results with one-line stubs (context editing; + * default: a quarter of the window; 0 = never). + */ + clearToolResultsAfterTokens?: number + /** Most recent tool results kept verbatim when clearing (default 3). */ + keepToolResults?: number +} + +export interface ResolvedCompaction { + auto: boolean + contextWindowTokens: number + thresholdTokens: number + keepRecentTurns: number + keepRecentSteps: number + summaryMaxTokens: number + maxToolOutputChars: number + clearToolResultsAfterTokens: number + keepToolResults: number +} + +export const resolveCompaction = (c: CompactionConfig | undefined): ResolvedCompaction => { + const contextWindowTokens = c?.contextWindowTokens ?? 128_000 + return { + auto: c?.auto ?? true, + contextWindowTokens, + thresholdTokens: c?.thresholdTokens ?? Math.floor(contextWindowTokens / 2), + keepRecentTurns: c?.keepRecentTurns ?? 4, + keepRecentSteps: c?.keepRecentSteps ?? 3, + summaryMaxTokens: c?.summaryMaxTokens ?? 1024, + maxToolOutputChars: c?.maxToolOutputChars ?? 20_000, + clearToolResultsAfterTokens: + c?.clearToolResultsAfterTokens ?? Math.floor(contextWindowTokens / 4), + keepToolResults: c?.keepToolResults ?? 3, + } +} + +/** Cheap, provider-independent token estimate (~4 chars per token). */ +export const estimateTokens = (text: string): number => Math.ceil((text ?? '').length / 4) + +const messageTokens = (msgs: StoredMessage[]): number => + msgs.reduce((n, m) => n + estimateTokens(`${m.role}: ${m.content}\n`), 0) + export interface CompressOptions { /** Compress once the transcript exceeds this many characters (0 = never). */ maxChars?: number + /** Compress once the transcript's estimated tokens exceed this (wins over maxChars). */ + thresholdTokens?: number /** How many most-recent messages to keep verbatim. */ keepRecent?: number + /** Output cap of the summary call. */ + summaryMaxTokens?: number + /** Compact even when below the threshold (manual "compact now"). */ + force?: boolean /** Time-box the summarisation call so it can never hang the caller. */ timeoutMs?: number abortSignal?: AbortSignal + /** Receives the summary call's token usage. */ + onUsage?: (usage: IUsage) => void } -const totalChars = (msgs: StoredMessage[]): number => msgs.reduce((n, m) => n + m.content.length, 0) +const SUMMARY_SYSTEM = + 'Summarize the following conversation into a compact set of durable facts, decisions, ' + + 'identifiers, open tasks and user preferences. Keep names, numbers and IDs verbatim. ' + + 'Output plain text only.' /** * When the transcript grows too long, summarise everything except the last @@ -25,26 +103,36 @@ export const compressHistory = async ( model: LanguageModel, opts: CompressOptions = {}, ): Promise => { - const maxChars = opts.maxChars ?? 12_000 const keepRecent = opts.keepRecent ?? 6 - if (maxChars <= 0 || messages.length <= keepRecent + 1 || totalChars(messages) <= maxChars) { - return messages + if (messages.length <= keepRecent + 1) return messages + if (!opts.force) { + if (opts.thresholdTokens !== undefined) { + if (opts.thresholdTokens <= 0 || messageTokens(messages) <= opts.thresholdTokens) { + return messages + } + } else { + const maxChars = opts.maxChars ?? 12_000 + const chars = messages.reduce((n, m) => n + m.content.length, 0) + if (maxChars <= 0 || chars <= maxChars) return messages + } } const head = messages.slice(0, messages.length - keepRecent) const tail = messages.slice(messages.length - keepRecent) + // The newest part of the head matters most; keep the tail end of it when it + // has to be cut to fit the summariser's own window. const transcript = head .map((m) => `${m.role}: ${m.content}`) .join('\n') - .slice(0, 8000) + .slice(-48_000) try { const result = await generate(model, { - system: - 'Summarize the following conversation into a compact set of durable facts and decisions. Output plain text only.', + system: SUMMARY_SYSTEM, prompt: transcript, - maxOutputTokens: 400, + maxOutputTokens: opts.summaryMaxTokens ?? 400, timeoutMs: opts.timeoutMs, abortSignal: opts.abortSignal, }) + opts.onUsage?.(normalizeUsage(result.usage)) const clean = (result.text || '').trim() if (!clean) return messages const summaryMsg: StoredMessage = { @@ -57,3 +145,43 @@ export const compressHistory = async ( return messages } } + +/** Alias with the sibling package's name; same behaviour as compressHistory. */ +export const compactHistory = compressHistory + +const TRACE_SYSTEM = + 'Summarize these completed agent steps into a compact progress note: what was done, ' + + 'what was found (keep names, numbers and IDs verbatim) and what failed. Plain text only.' + +/** + * Fold the oldest entries of a run's step log into one summary entry, keeping + * the last `keepRecent` verbatim. Returns the input unchanged when it fits the + * threshold or on any failure; never throws. + */ +export const compactSteps = async ( + done: string[], + model: LanguageModel, + opts: CompressOptions & { thresholdTokens: number }, +): Promise => { + const keepRecent = opts.keepRecent ?? 3 + if (done.length <= keepRecent + 1) return done + if (!opts.force && estimateTokens(done.join('\n')) <= opts.thresholdTokens) return done + const head = done.slice(0, done.length - keepRecent) + try { + const result = await generate(model, { + system: TRACE_SYSTEM, + prompt: head.map((d, i) => `${i + 1}. ${d}`).join('\n'), + maxOutputTokens: opts.summaryMaxTokens ?? 600, + timeoutMs: opts.timeoutMs, + abortSignal: opts.abortSignal, + }) + opts.onUsage?.(normalizeUsage(result.usage)) + const clean = (result.text || '').trim() + if (!clean) return done + return [`Summary of ${head.length} earlier step(s): ${clean}`, ...done.slice(head.length)] + } catch { + return done + } +} + +export { messageTokens as estimateMessagesTokens } diff --git a/src/parse.ts b/src/parse.ts index 22d4026..061537a 100644 --- a/src/parse.ts +++ b/src/parse.ts @@ -14,7 +14,15 @@ export interface RawAction { export type ReplanDecision = 'continue' | 'revise' | 'finish' -const stripFences = (text: string): string => (text || '').replace(/```(?:json)?/gi, '').trim() +// Reasoning models running locally (Qwen3, DeepSeek-R1 distills) put their +// chain of thought inline in tags; it is never part of the answer. +const stripThinking = (text: string): string => + text.replace(/[\s\S]*?<\/think>/gi, '').replace(/^[\s\S]*?<\/think>/i, '') + +const stripFences = (text: string): string => + stripThinking(text || '') + .replace(/```(?:json)?/gi, '') + .trim() /** Extracts the first balanced {...} or [...] block, respecting strings. */ const extractBalanced = (text: string, open: '{' | '['): string | undefined => { @@ -128,6 +136,8 @@ const salvageString = (text: string, field: string): string => { export interface PlannerResult { reply: string plan: string[] + /** Skill names the planner picked (empty when none / not asked). */ + skills: string[] } export const parsePlannerResponse = (raw: string): PlannerResult => { @@ -135,15 +145,16 @@ export const parsePlannerResponse = (raw: string): PlannerResult => { const obj = asObject(text) if (obj) { const reply = typeof obj.reply === 'string' ? obj.reply.trim() : '' - return { reply, plan: normalizeSteps(obj.plan) } + return { reply, plan: normalizeSteps(obj.plan), skills: normalizeSteps(obj.skills) } } if (/"plan"\s*:/.test(text)) { return { reply: salvageString(text, 'reply'), plan: normalizeSteps(salvageStringArray(text, 'plan')), + skills: normalizeSteps(salvageStringArray(text, 'skills')), } } - return { reply: looksLikeJson(text) ? '' : text.trim(), plan: [] } + return { reply: looksLikeJson(text) ? '' : text.trim(), plan: [], skills: [] } } // --- executor (tool calls) ------------------------------------------------- diff --git a/src/prompts.ts b/src/prompts.ts index 07cf3f4..a03ad8f 100644 --- a/src/prompts.ts +++ b/src/prompts.ts @@ -8,6 +8,15 @@ * out an exact JSON shape and (for the executor) include a tool catalogue; * the output is salvaged by parse.ts. * + * Layout matters for prompt caching: the SYSTEM part holds only what is stable + * for a whole run (role, skills, the planner's tool catalogue), the user + * prompt holds everything that changes (goal, state, history, progress), so a + * provider's prefix cache can reuse the system part across calls. + * + * The agent is autonomous by design: it plans tool use for anything the tools + * can do or look up, never asks the user for confirmation (consent is the + * host's `toolApproval` policy), and states assumptions instead of asking. + * * Override any builder via `BrowserAgentConfig.prompts`. */ @@ -27,6 +36,10 @@ export interface PlannerPromptContext { mode: ToolCallMode /** Prior session messages (oldest first), for resolving references to earlier turns. */ history?: { role: string; content: string }[] + /** "- name: description" index of the configured skills, when any. */ + skills?: string + /** The catalogue is condensed and the executor can search the rest (large catalogues). */ + searchMode?: boolean } export interface ExecutorPromptContext { goal: string @@ -35,8 +48,15 @@ export interface ExecutorPromptContext { index: number total: number toolCatalog: string + /** Earlier steps of this run, each with its outcome and key data. */ done: string[] mode: ToolCallMode + /** "- name: description" index of the configured skills, when any. */ + skills?: string + /** Full instructions of the skills active in this run. */ + activeSkills?: string + /** The executor can call find_tools to activate more tools. */ + searchMode?: boolean } export interface ReplannerPromptContext { goal: string @@ -44,11 +64,15 @@ export interface ReplannerPromptContext { done: string[] remaining: string[] mode: ToolCallMode + activeSkills?: string } export interface SynthesizerPromptContext { goal: string state?: string done: string[] + /** Excerpts of what the tools returned, for answering with real data. */ + findings?: string[] + activeSkills?: string } export interface Prompts { @@ -71,64 +95,99 @@ const historyBlock = (history?: { role: string; content: string }[]): string => return `\n\nCONVERSATION SO FAR:\n${lines.join('\n')}` } +const section = (title: string, body?: string): string => + body && body.trim() ? `\n\n${title}:\n${body.trim()}` : '' + // --- planner --------------------------------------------------------------- -const PLANNER_NATIVE = `You are the PLANNER of a tool-using agent that changes a workspace step by step. -Produce a brief "thought" and an ordered "steps" list (1–6 DISTINCT, self-contained steps) grounded in the current STATE and the available TOOLS. -If the user only greets, thanks, makes small talk, asks a question, or is unclear: return an EMPTY steps list and put a short, friendly answer in "thought". -If a CONVERSATION SO FAR section is present, use it to resolve references to earlier turns. -Never repeat or pad steps.` +const PLANNER_RULES = `- Act autonomously: whenever the TOOLS can do the work or look up the answer, plan the steps — a question that needs data, a lookup or a check is a real goal too. +- Never plan a step that asks the user something. If the request is ambiguous, pick the most reasonable interpretation and say which one you chose. +- Plan realistic steps the available TOOLS can perform, grounded in the current STATE. +- If a CONVERSATION SO FAR section is present, use it to resolve references to earlier turns. +- Never repeat or pad steps.` + +const PLANNER_NATIVE = `You are the PLANNER of an autonomous tool-using agent that works step by step. +Produce a brief "thought" and an ordered "steps" list (1–6 DISTINCT, self-contained steps). +Return an EMPTY steps list ONLY for a pure greeting, thanks or small talk, or a question you can answer completely from the STATE and general knowledge without any tool — then put the answer itself in "thought". +${PLANNER_RULES}` -const PLANNER_PROMPTED = `You are the PLANNER of a tool-using agent that changes a workspace step by step. +const PLANNER_PROMPTED = `You are the PLANNER of an autonomous tool-using agent that works step by step. Reply with a single JSON object, nothing else: -{ "reply": string, "plan": string[] } -- Only produce a "plan" when the user CLEARLY asks to build, do, or change something. For a greeting, small talk, thanks, a question, or an unclear/empty request: set "plan": [] and put a short, friendly "reply". -- For a real goal: "reply" is one short sentence; "plan" is 1–6 DISTINCT, self-contained steps. Never repeat or pad steps. -- Plan realistic steps the available TOOLS can perform, grounded in the current STATE. -- If a CONVERSATION SO FAR section is present, use it to resolve references to earlier turns.` +{ "reply": string, "plan": string[], "skills": string[] } +- For a real goal: "reply" is one short sentence; "plan" is 1–6 DISTINCT, self-contained steps. +- Set "plan": [] ONLY for a pure greeting, thanks or small talk, or a question you can answer completely from the STATE and general knowledge without any tool — then put the answer itself in "reply". +${PLANNER_RULES}` + +const PLANNER_SKILLS = `\nIf a SKILLS list is present, list the names of the skills that apply to this goal in "skills" (or leave it empty).` + +const PLANNER_SEARCH = `\nThe TOOLS list may be abbreviated: the executor can search the full catalogue, so describe WHAT to do; name tools only when you see them listed.` // --- executor -------------------------------------------------------------- -const EXECUTOR_NATIVE = `You are the EXECUTOR of a tool-using agent. Carry out ONLY the current step by calling the provided tools. -Build on the current STATE — do not repeat work that is already there. -When the step is done, reply with one short human sentence describing what you did (no JSON). -If you CANNOT complete the step (a needed tool is missing or an input is unavailable), explain why in one sentence and include the token [BLOCKER].` +const EXECUTOR_RULES = `- Work autonomously: never ask the user questions or for confirmation. Look things up with the tools, choose sensible defaults, and state any assumption in your reply. Permission to run tools is handled by the system, not by you. +- Build on the current STATE and on the results of earlier steps — do not repeat work that is already done.` + +const EXECUTOR_NATIVE = `You are the EXECUTOR of an autonomous tool-using agent. Carry out ONLY the current step by calling the provided tools. +${EXECUTOR_RULES} +- When the step is done, reply with a short factual summary of what you did and found, including the concrete data (names, ids, numbers, values) later steps or the final answer need. No JSON. +- If you truly CANNOT complete the step (a needed tool is missing or was denied, credentials or data are unavailable), explain why in one sentence and include the token [BLOCKER].` -const EXECUTOR_PROMPTED = `You are the EXECUTOR of a tool-using agent. Carry out ONLY the current step by emitting tool calls. +const EXECUTOR_PROMPTED = `You are the EXECUTOR of an autonomous tool-using agent. Carry out ONLY the current step by emitting tool calls. Reply with a single JSON object, nothing else: { "reply": string, "actions": [ { "tool": string, "args": object } ] } - "actions" are the tool calls for THIS step ([] if none are needed). Use ONLY tools from the TOOLS list; "args" must match the tool's parameters. -- Build on the current STATE — do not repeat work that is already there. -- "reply" is one short human sentence (no JSON) describing what you did. +${EXECUTOR_RULES} +- "reply" is a short factual summary (no JSON) of what you did and found, with the concrete data later steps need. - After your actions run you will see their TOOL RESULTS and may continue the same step; finish with "actions": [] once it is done. -- If you CANNOT complete the step, set "actions": [] and put the token [BLOCKER] in "reply" with a short reason.` +- If you truly CANNOT complete the step, set "actions": [] and put the token [BLOCKER] in "reply" with a short reason.` + +const EXECUTOR_SEARCH = `\n- If a tool you need is not in your current tool list, call find_tools with keywords; the tools it returns become callable right after.` + +const EXECUTOR_SKILLS = `\n- If a skill from the SKILLS list covers this step, call load_skill to read it first (read_skill_file for its bundled files).` // --- replanner ------------------------------------------------------------- -const REPLANNER_NATIVE = `You are the REPLANNER. After an executed step, decide whether to keep going, revise the remaining steps, or finish. +const REPLANNER_RULES = `Judge from the current STATE and the progress vs the goal. Prefer "continue". +After a failure, prefer revising around it (another tool, another approach) over giving up; never add a step that asks the user something.` + +const REPLANNER_NATIVE = `You are the REPLANNER of an autonomous agent. After an executed step, decide whether to keep going, revise the remaining steps, or finish. "continue": the remaining steps still fit. "revise": provide a better "plan" for the REMAINING work (never repeat done work). "finish": the goal is already met. -Judge from the current STATE vs the goal. Prefer "continue".` +${REPLANNER_RULES}` -const REPLANNER_PROMPTED = `You are the REPLANNER. After each executed step you decide whether to keep going, revise the remaining steps, or finish. +const REPLANNER_PROMPTED = `You are the REPLANNER of an autonomous agent. After each executed step you decide whether to keep going, revise the remaining steps, or finish. Reply with a single JSON object, nothing else: { "decision": "continue" | "revise" | "finish", "reason": string, "plan": string[] } - "continue": remaining steps still fit — proceed (omit "plan"). - "revise": replace the REMAINING steps with a better list in "plan"; never repeat done work. - "finish": the goal is already met — stop (omit "plan"). -Judge from the current STATE vs the goal. Prefer "continue".` +${REPLANNER_RULES}` -const SYNTHESIZER_SYSTEM = `You are the SYNTHESIZER. In 1–3 short, friendly sentences, tell the user what was done to meet their goal. Be concrete. Output plain text only — no JSON, no code.` +const SYNTHESIZER_SYSTEM = `You are the SYNTHESIZER. Answer the user's goal from what the agent did and found. +- Lead with the result: the concrete data, from the FINDINGS verbatim when accuracy matters (names, numbers, ids), then — briefly — what was changed. +- Mention any assumption the agent made. If something could not be done, say so plainly and why. +- Do not ask follow-up questions unless the goal genuinely cannot be completed without the user. +- Be concise and friendly. Plain text (light markdown is fine) — no JSON, no code.` export const defaultPrompts: Prompts = { planner: (ctx) => ({ - system: ctx.mode === 'prompted' ? PLANNER_PROMPTED : PLANNER_NATIVE, - prompt: `GOAL: ${ctx.goal}${historyBlock(ctx.history)}${stateBlock(ctx.state)}\n\nTOOLS:\n${ctx.toolCatalog}`, + system: + (ctx.mode === 'prompted' ? PLANNER_PROMPTED : PLANNER_NATIVE) + + (ctx.skills ? PLANNER_SKILLS : '') + + (ctx.searchMode ? PLANNER_SEARCH : '') + + section('SKILLS', ctx.skills) + + `\n\nTOOLS:\n${ctx.toolCatalog}`, + prompt: `GOAL: ${ctx.goal}${historyBlock(ctx.history)}${stateBlock(ctx.state)}`, }), executor: (ctx) => ({ - system: ctx.mode === 'prompted' ? EXECUTOR_PROMPTED : EXECUTOR_NATIVE, + system: + (ctx.mode === 'prompted' ? EXECUTOR_PROMPTED : EXECUTOR_NATIVE) + + (ctx.searchMode ? EXECUTOR_SEARCH : '') + + (ctx.skills ? EXECUTOR_SKILLS : '') + + section('SKILLS', ctx.skills) + + section('ACTIVE SKILL INSTRUCTIONS', ctx.activeSkills), prompt: [ `GOAL: ${ctx.goal}`, stateBlock(ctx.state).trimStart(), @@ -140,7 +199,9 @@ export const defaultPrompts: Prompts = { .join('\n\n'), }), replanner: (ctx) => ({ - system: ctx.mode === 'prompted' ? REPLANNER_PROMPTED : REPLANNER_NATIVE, + system: + (ctx.mode === 'prompted' ? REPLANNER_PROMPTED : REPLANNER_NATIVE) + + section('ACTIVE SKILL INSTRUCTIONS', ctx.activeSkills), prompt: [ `GOAL: ${ctx.goal}`, stateBlock(ctx.state).trimStart(), @@ -152,12 +213,13 @@ export const defaultPrompts: Prompts = { .join('\n\n'), }), synthesizer: (ctx) => ({ - system: SYNTHESIZER_SYSTEM, + system: SYNTHESIZER_SYSTEM + section('ACTIVE SKILL INSTRUCTIONS', ctx.activeSkills), prompt: [ `GOAL: ${ctx.goal}`, stateBlock(ctx.state).trimStart(), `What was done:\n${numbered(ctx.done, '(no changes)')}`, - 'Write the final summary for the user.', + ctx.findings?.length ? `FINDINGS (tool results):\n${ctx.findings.join('\n')}` : '', + 'Write the final answer for the user.', ] .filter(Boolean) .join('\n\n'), diff --git a/src/providers/capabilities.ts b/src/providers/capabilities.ts index 5201a5d..2817948 100644 --- a/src/providers/capabilities.ts +++ b/src/providers/capabilities.ts @@ -35,3 +35,48 @@ export const supportsStructuredOutput = (p: ProviderType): boolean => CLOUD.has( */ export const directBrowserOk = (p: ProviderType): boolean => p === 'google' || p === 'gateway' || p === 'openai-compatible' || p === 'anthropic' + +// Local / on-device runtimes: the prompted path is text-only. +const LOCAL_RE = /web-?llm|mlc|browser-ai|transformers|built-?in/i +// Text-only model families under otherwise vision-capable providers. +const TEXT_ONLY_MODEL_RE = + /gpt-3\.5|o1-mini|o3-mini|deepseek|text-(davinci|embedding)|embedding|babbage|davinci|codestral/i +const VISION_PROVIDER_RE = /^(google|anthropic|openai|azure|xai|vertex|bedrock|gemini)/i +// Providers that read PDFs natively (as file parts). +const PDF_PROVIDER_RE = /^(google|anthropic|openai|azure|vertex|bedrock|gemini)/i + +/** + * Whether a model accepts image input: `true` (expected to), `false` (known + * not to — local runtimes, DeepSeek, text-only families), or `undefined` + * (unknown, e.g. an OpenAI-compatible server — the agent tries and turns a + * provider refusal into a clear error). Override per agent with `vision`. + */ +export const supportsImages = (model: unknown): boolean | undefined => { + if (typeof model === 'string') { + // A gateway model id: "provider/model". + const [provider = '', id = model] = model.includes('/') ? model.split('/', 2) : ['', model] + if (TEXT_ONLY_MODEL_RE.test(id) || /deepseek/i.test(provider)) return false + return VISION_PROVIDER_RE.test(provider) ? true : undefined + } + if (!model || typeof model !== 'object') return undefined + const provider = String((model as { provider?: unknown }).provider ?? '') + const id = String((model as { modelId?: unknown }).modelId ?? '') + if (LOCAL_RE.test(provider)) return false + if (/deepseek/i.test(provider) || TEXT_ONLY_MODEL_RE.test(id)) return false + return VISION_PROVIDER_RE.test(provider) ? true : undefined +} + +/** + * Whether a model reads PDF files natively: `false` for local runtimes and + * text-only families, `true` for Gemini, Claude and OpenAI's vision models, + * `undefined` when unknown (tried; a refusal becomes a clear error). + */ +export const supportsPdf = (model: unknown): boolean | undefined => { + const vision = supportsImages(model) + if (vision === false) return false + const provider = + typeof model === 'string' + ? (model.split('/')[0] ?? '') + : String((model as { provider?: unknown } | undefined)?.provider ?? '') + return PDF_PROVIDER_RE.test(provider) ? true : undefined +} diff --git a/src/skills.ts b/src/skills.ts new file mode 100644 index 0000000..05e3747 --- /dev/null +++ b/src/skills.ts @@ -0,0 +1,211 @@ +import { jsonSchema, tool, type ToolSet } from 'ai' +import { clip } from './llm/util.js' + +/** + * Skills — reusable instruction bundles in the agentskills.io shape + * (`SKILL.md` with `name`/`description` frontmatter, a markdown body, optional + * bundled text files). Progressive disclosure: only the one-line index sits in + * the prompts; a skill's full instructions enter the context when the planner + * selects it or the executor loads it with `load_skill`. + */ +export interface Skill { + /** Stable id: lowercase letters, digits and dashes, 1–64 chars. */ + name: string + /** When to use it — the only part always shown to the model. */ + description: string + /** Full instructions (the markdown body of SKILL.md). */ + content: string + /** Bundled text resources, addressed by skill-relative POSIX paths. */ + files?: { path: string; content: string }[] +} + +export const SKILL_NAME_RE = /^[a-z0-9][a-z0-9-]{0,63}$/ + +/** Names of the built-in skill tools (reserved when skills are configured). */ +export const SKILL_TOOL_NAMES = ['load_skill', 'read_skill_file'] as const + +/** Validate a skill definition and return it (throws a precise error). */ +export const defineSkill = (skill: Skill): Skill => { + if (!skill || typeof skill !== 'object') throw new Error('skill must be an object') + if (typeof skill.name !== 'string' || !SKILL_NAME_RE.test(skill.name)) { + throw new Error( + `skill name ${JSON.stringify(skill.name)} is invalid: use 1-64 lowercase letters, digits and dashes`, + ) + } + if (typeof skill.description !== 'string' || !skill.description.trim()) { + throw new Error(`skill "${skill.name}" needs a description`) + } + if (typeof skill.content !== 'string') { + throw new Error(`skill "${skill.name}" needs a content string`) + } + for (const f of skill.files ?? []) { + if (typeof f?.path !== 'string' || !f.path || f.path.startsWith('/') || f.path.includes('..')) { + throw new Error(`skill "${skill.name}": invalid file path ${JSON.stringify(f?.path)}`) + } + } + return skill +} + +const unquote = (v: string): string => { + const t = v.trim() + if (t.length >= 2 && t.startsWith('"') && t.endsWith('"')) { + try { + return JSON.parse(t) as string + } catch { + return t.slice(1, -1) + } + } + if (t.length >= 2 && t.startsWith("'") && t.endsWith("'")) { + return t.slice(1, -1).replace(/''/g, "'") + } + return t +} + +/** + * Parse the small YAML subset SKILL.md frontmatter uses: `key: value` pairs, + * single/double-quoted values, and `>` / `|` block scalars. Unknown keys are + * kept (as strings) but ignored by the agent. + */ +const parseFrontmatter = (block: string): Record => { + const out: Record = {} + const lines = block.split(/\r?\n/) + for (let i = 0; i < lines.length; i += 1) { + const m = /^([A-Za-z0-9_-]+):\s*(.*)$/.exec(lines[i]) + if (!m) continue + const [, key, rest] = m + if (rest === '>' || rest === '|' || rest === '>-' || rest === '|-') { + const body: string[] = [] + while (i + 1 < lines.length && (/^\s+\S/.test(lines[i + 1]) || lines[i + 1] === '')) { + body.push(lines[i + 1].trim()) + i += 1 + } + out[key] = rest.startsWith('>') ? body.filter(Boolean).join(' ') : body.join('\n').trim() + continue + } + out[key] = unquote(rest) + } + return out +} + +/** + * Parse a SKILL.md document into a {@link Skill}. Frontmatter must provide + * `name` and `description`; the markdown body becomes `content`. + */ +export const parseSkillMarkdown = ( + markdown: string, + files?: { path: string; content: string }[], +): Skill => { + const text = (markdown ?? '').replace(/^/, '') + const m = /^---\r?\n([\s\S]*?)\r?\n---\r?\n?([\s\S]*)$/.exec(text) + if (!m) throw new Error('SKILL.md must start with a --- frontmatter block (name, description)') + const meta = parseFrontmatter(m[1]) + return defineSkill({ + name: meta.name ?? '', + description: meta.description ?? '', + content: m[2].trim(), + ...(files?.length ? { files } : {}), + }) +} + +export interface LoadSkillOptions { + /** Bundled files to fetch, relative to the SKILL.md URL. */ + files?: string[] + fetch?: typeof fetch +} + +/** Fetch a SKILL.md (and optionally its bundled files) over HTTP. */ +export const loadSkillFromUrl = async ( + url: string, + opts: LoadSkillOptions = {}, +): Promise => { + const fetchFn = opts.fetch ?? globalThis.fetch + const res = await fetchFn(url) + if (!res.ok) throw new Error(`could not fetch ${url}: HTTP ${res.status}`) + const markdown = await res.text() + const files: { path: string; content: string }[] = [] + for (const path of opts.files ?? []) { + const r = await fetchFn(new URL(path, url).toString()) + if (!r.ok) throw new Error(`could not fetch skill file ${path}: HTTP ${r.status}`) + files.push({ path, content: await r.text() }) + } + return parseSkillMarkdown(markdown, files) +} + +/** "- name: description" lines for the prompts. */ +export const renderSkillIndex = (skills: Skill[]): string => + skills.map((s) => `- ${s.name}: ${s.description.replace(/\s+/g, ' ').trim()}`).join('\n') + +const ACTIVE_SKILLS_BUDGET = 24_000 + +/** The full instructions of the active skills, within a total character budget. */ +export const renderActiveSkills = (skills: Skill[]): string => { + let budget = ACTIVE_SKILLS_BUDGET + const parts: string[] = [] + for (const s of skills) { + if (budget <= 0) break + const files = s.files?.length + ? `\n(bundled files, read with read_skill_file: ${s.files.map((f) => f.path).join(', ')})` + : '' + const block = clip(`### Skill: ${s.name}\n${s.content}${files}`, budget) + budget -= block.length + parts.push(block) + } + return parts.join('\n\n') +} + +/** + * The built-in `load_skill` / `read_skill_file` tools. `onLoad` fires when a + * skill is loaded so the runner can keep it active for later steps. + */ +export const createSkillTools = (skills: Skill[], onLoad?: (skill: Skill) => void): ToolSet => { + const byName = new Map(skills.map((s) => [s.name, s])) + const known = skills.map((s) => s.name).join(', ') + const loadSkill = tool({ + description: + 'Load the full instructions of a skill from the SKILLS list before doing work it covers.', + inputSchema: jsonSchema<{ name: string }>({ + type: 'object', + properties: { name: { type: 'string', description: 'The skill name' } }, + required: ['name'], + additionalProperties: false, + }), + execute: async ({ name }) => { + const skill = byName.get(name) + if (!skill) throw new Error(`unknown skill "${name}"; available: ${known}`) + onLoad?.(skill) + return { + name: skill.name, + content: skill.content, + files: (skill.files ?? []).map((f) => f.path), + } + }, + }) + const readFile = tool({ + description: "Read one of a skill's bundled files (paths are listed by load_skill).", + inputSchema: jsonSchema<{ name: string; path: string }>({ + type: 'object', + properties: { + name: { type: 'string', description: 'The skill name' }, + path: { type: 'string', description: 'The file path, relative to the skill' }, + }, + required: ['name', 'path'], + additionalProperties: false, + }), + execute: async ({ name, path }) => { + const skill = byName.get(name) + if (!skill) throw new Error(`unknown skill "${name}"; available: ${known}`) + const file = skill.files?.find((f) => f.path === path) + if (!file) { + const listed = (skill.files ?? []).map((f) => f.path).join(', ') || 'none' + throw new Error(`skill "${name}" has no file "${path}"; files: ${listed}`) + } + return file.content + }, + }) + for (const t of [loadSkill, readFile]) { + ;(t as { readOnly?: boolean }).readOnly = true + ;(t as { promptHint?: string }).promptHint = + t === loadSkill ? '{ name: string }' : '{ name: string, path: string }' + } + return { load_skill: loadSkill, read_skill_file: readFile } +} diff --git a/src/storage/db.ts b/src/storage/db.ts index 6ad9ae1..7ef4c29 100644 --- a/src/storage/db.ts +++ b/src/storage/db.ts @@ -8,6 +8,10 @@ export interface AgentWebDBOptions { export const KEYS_STORE = 'keys' export const SECRETS_STORE = 'secrets' export const SESSIONS_STORE = 'sessions' +export const FILES_STORE = 'files' + +// v1: keys, secrets, sessions. v2: + files (the virtual file system). +const DB_VERSION = 2 let cache = new Map>() @@ -22,11 +26,13 @@ export const openAgentWebDB = (opts: AgentWebDBOptions = {}): Promise + proxied: ProxiedToolSpec[] + } + | { type: 'abort' } + | { type: 'tool-result'; callId: string; ok: boolean; output: unknown } + +export type WorkerToParent = + | { type: 'event'; event: AgentEvent } + | { type: 'tool-call'; callId: string; name: string; input: unknown } + | { type: 'done'; text: string; usage: IUsage; steps: number } + | { type: 'error'; error: string } + +/** + * The slice of the Worker / DedicatedWorkerGlobalScope / MessagePort API we + * use. Structural, so a real `Worker`, a worker's `self`, or a MessageChannel + * port (tests) all fit. + */ +export interface MessageEndpoint { + postMessage(message: unknown): void + addEventListener(type: string, listener: (ev: any) => void): void + removeEventListener?(type: string, listener: (ev: any) => void): void +} + +/** A worker as the parent sees it. */ +export interface WorkerLike extends MessageEndpoint { + terminate?(): void +} + +/** JSON-safe copy (Errors become their message, cycles become strings). */ +export const toCloneable = (value: unknown): unknown => { + try { + return JSON.parse( + JSON.stringify(value, (_k, v: unknown) => (v instanceof Error ? v.message : v)) ?? 'null', + ) + } catch { + return String(value) + } +} + +/** postMessage that never throws a DataCloneError: falls back to a JSON-safe copy. */ +export const safePost = (target: { postMessage(m: unknown): void }, message: unknown): void => { + try { + target.postMessage(message) + } catch { + target.postMessage(toCloneable(message)) + } +} diff --git a/src/subagent/tool.ts b/src/subagent/tool.ts new file mode 100644 index 0000000..270f706 --- /dev/null +++ b/src/subagent/tool.ts @@ -0,0 +1,354 @@ +import { asSchema, jsonSchema, tool, type Tool, type ToolSet } from 'ai' +import type { IUsage } from '../agent/loop-types.js' +import { createAgent, type Agent } from '../agent/runner.js' +import type { BrowserAgentConfig } from '../config.js' +import type { AgentEvent } from '../events.js' +import { clip } from '../llm/util.js' +import type { ProviderModelSpec } from '../providers/types.js' +import type { CredentialStore } from '../secrets/store.js' +import { isReadOnlyTool } from '../tools/approval.js' +import type { AgentTool } from '../tools/types.js' +import { toolRunContextOf } from '../tools/wrap.js' +import { + safePost, + toCloneable, + type ProxiedToolSpec, + type WorkerLike, + type WorkerToParent, +} from './protocol.js' + +/** The serializable slice of BrowserAgentConfig a worker subagent can be built from. */ +export type SubagentWorkerConfig = Pick< + BrowserAgentConfig, + | 'clientName' + | 'systemPrompt' + | 'availableTools' + | 'excludedTools' + | 'toolMode' + | 'toolSelectionStrategy' + | 'toolSearchThreshold' + | 'skills' + | 'maxIterations' + | 'maxStepsPerTask' + | 'maxRevisions' + | 'maxToolCalls' + | 'maxPlanSteps' + | 'chatTimeoutMs' + | 'budgets' + | 'limits' + | 'temperature' + | 'thinking' + | 'stageThinking' + | 'promptCaching' + | 'compaction' + | 'replan' + | 'synthesize' + | 'logLevel' +> & { + /** The child's model, resolved inside the worker (by its `resolveModel`, or the registry). */ + model: ProviderModelSpec + /** Consent policy inside the worker — mode and rules only (no callbacks cross threads). */ + toolApproval?: { mode?: 'autopilot' | 'read-only'; rules?: Record } +} + +export interface SubagentToolOptions { + /** Label for events and logs (the tool's KEY in your ToolSet is up to you). */ + name: string + /** What the subagent is for — the parent model reads this to decide when to delegate. */ + description: string + /** In-process subagent: a full config (models, tools, …) for the child agent. */ + config?: BrowserAgentConfig + /** + * Worker subagent: builds a fresh Worker per delegated task, e.g. + * `() => new Worker(new URL('./agent.worker.ts', import.meta.url), { type: 'module' })`. + * The worker module calls `serveSubagentWorker()`. + */ + worker?: () => WorkerLike + /** Worker subagent: the child's serializable config. */ + workerConfig?: SubagentWorkerConfig + /** + * Resolves `workerConfig.model.credentialRef` in the main thread; the key is + * posted to the worker for that one task and never stored there. + */ + credentials?: CredentialStore + /** + * Host tools the subagent may call. In-process they are merged into the + * child's tools; with a worker they stay in the main thread and are called + * over RPC. Either way each call passes the PARENT's consent gate. + */ + tools?: ToolSet + /** Parallel delegations allowed at once (default 4); extra calls queue. */ + maxConcurrent?: number + /** Give up on one delegated task after this many ms (default: none). */ + timeoutMs?: number + /** Cap on the text handed back to the parent model (default 8000 chars). */ + outputMaxChars?: number + /** Mark the tool read-only for the parent's consent gate (default false). */ + readOnly?: boolean +} + +interface ChildOutcome { + text: string + usage: IUsage +} + +const abortError = (): Error => { + const e = new Error('The subagent was aborted') + e.name = 'AbortError' + return e +} + +/** A tiny FIFO semaphore: at most `size` holders, abort-aware waiting. */ +const createSemaphore = (size: number) => { + let active = 0 + const queue: (() => void)[] = [] + return async (signal?: AbortSignal): Promise<() => void> => { + if (active >= size) { + await new Promise((resolve, reject) => { + const go = () => { + signal?.removeEventListener('abort', onAbort) + resolve() + } + const onAbort = () => { + const i = queue.indexOf(go) + if (i >= 0) queue.splice(i, 1) + reject(abortError()) + } + queue.push(go) + signal?.addEventListener('abort', onAbort, { once: true }) + }) + } + active += 1 + let released = false + return () => { + if (released) return + released = true + active -= 1 + queue.shift()?.() + } + } +} + +let seq = 0 + +/** + * Expose a subagent as a tool: the parent delegates `{ task }`, the child runs + * its own plan → execute → synthesize loop — in this thread, or isolated in a + * Web Worker — and its final answer comes back as the tool result. Several + * delegations in one model step run in parallel (bounded by `maxConcurrent`). + * + * The parent sees `subagent.start` / `subagent.event` / `subagent.complete` / + * `subagent.error` events, and the child's tokens are charged to the parent + * run (a `usage` event with phase 'subagent'), so the parent's limits apply. + */ +export const createSubagentTool = (opts: SubagentToolOptions): AgentTool => { + if (!opts.config === !opts.worker) { + throw new Error( + `subagent "${opts.name}": pass exactly one of \`config\` (in-process) or \`worker\``, + ) + } + if (opts.worker && !opts.workerConfig) { + throw new Error(`subagent "${opts.name}": a worker subagent needs \`workerConfig\``) + } + const acquire = createSemaphore(Math.max(1, opts.maxConcurrent ?? 4)) + const outputMax = opts.outputMaxChars ?? 8_000 + + const gatedHostTools = ( + approve: ((name: string, input: unknown, readOnly: boolean) => Promise) | undefined, + ): ToolSet => + Object.fromEntries( + Object.entries(opts.tools ?? {}).map(([name, t]) => { + const execute = (t as { execute?: (i: unknown, o: unknown) => unknown }).execute + if (typeof execute !== 'function' || !approve) return [name, t] + return [ + name, + { + ...t, + execute: async (input: unknown, o: unknown) => { + await approve(name, input, isReadOnlyTool(t as Tool)) + return execute.call(t, input, o) + }, + }, + ] + }), + ) + + const runInProcess = async ( + task: string, + signal: AbortSignal | undefined, + forward: (e: AgentEvent) => void, + hostTools: ToolSet, + ): Promise => { + const config = opts.config as BrowserAgentConfig + // One child per task: its host tools are gated by THIS parent run. A + // direct model (incl. a loaded WebLLM engine) is passed through, not rebuilt. + const agent: Agent = await createAgent({ ...config, tools: { ...config.tools, ...hostTools } }) + let runError: string | undefined + const result = await agent.run(task, { + signal, + onEvent: (e) => { + if (e.type === 'error' && e.phase === 'run') runError = e.error + forward(e) + }, + }) + if (result.stopped) throw abortError() + if (runError) throw new Error(runError) + return { text: result.final, usage: result.usage } + } + + const proxiedSpecs = async (): Promise => + Promise.all( + Object.entries(opts.tools ?? {}).map(async ([name, t]) => ({ + name, + description: typeof t.description === 'string' ? t.description : '', + inputSchema: toCloneable(await asSchema(t.inputSchema as never).jsonSchema), + readOnly: isReadOnlyTool(t as Tool), + })), + ) + + const workerConfigFor = async (): Promise> => { + const cfg = opts.workerConfig as SubagentWorkerConfig + let model: ProviderModelSpec = { ...cfg.model } + if (model.credentialRef && !model.apiKey && opts.credentials) { + const apiKey = await opts.credentials.getApiKey(model.credentialRef) + if (!apiKey) + throw new Error(`no API key stored under "${model.credentialRef}" for the subagent`) + model = { ...model, apiKey } + } + return toCloneable({ ...cfg, model }) as Record + } + + const runInWorker = async ( + task: string, + signal: AbortSignal | undefined, + forward: (e: AgentEvent) => void, + hostTools: ToolSet, + ): Promise => { + if (signal?.aborted) throw abortError() + const [config, proxied] = await Promise.all([workerConfigFor(), proxiedSpecs()]) + const worker = (opts.worker as () => WorkerLike)() + return new Promise((resolve, reject) => { + let settled = false + const finish = (fn: () => void): void => { + if (settled) return + settled = true + signal?.removeEventListener('abort', onAbort) + worker.removeEventListener?.('message', onMessage) + worker.removeEventListener?.('error', onError) + // A worker is single-use: terminating frees its memory and any engine. + worker.terminate?.() + fn() + } + const onAbort = (): void => { + safePost(worker, { type: 'abort' }) + finish(() => reject(abortError())) + } + const onError = (ev: { message?: string }): void => + finish(() => reject(new Error(`subagent worker failed: ${ev?.message ?? 'unknown error'}`))) + const callHostTool = async (callId: string, name: string, input: unknown): Promise => { + const t = hostTools[name] as { execute?: (i: unknown, o: unknown) => unknown } | undefined + try { + if (!t || typeof t.execute !== 'function') throw new Error(`unknown host tool "${name}"`) + const output = await t.execute(input, { + toolCallId: callId, + messages: [], + abortSignal: signal, + }) + if (!settled) safePost(worker, { type: 'tool-result', callId, ok: true, output }) + } catch (err) { + const output = err instanceof Error ? err.message : String(err) + if (!settled) safePost(worker, { type: 'tool-result', callId, ok: false, output }) + } + } + const onMessage = (ev: { data: unknown }): void => { + const msg = ev.data as WorkerToParent + switch (msg?.type) { + case 'event': + forward(msg.event) + break + case 'tool-call': + void callHostTool(msg.callId, msg.name, msg.input) + break + case 'done': + finish(() => resolve({ text: msg.text, usage: msg.usage })) + break + case 'error': + finish(() => reject(new Error(msg.error))) + break + } + } + worker.addEventListener('message', onMessage) + worker.addEventListener('error', onError) + signal?.addEventListener('abort', onAbort, { once: true }) + safePost(worker, { type: 'run', task, config, proxied }) + }) + } + + const t = tool({ + description: opts.description, + inputSchema: jsonSchema<{ task: string }>({ + type: 'object', + properties: { + task: { + type: 'string', + description: + 'A complete, self-contained task: the subagent sees nothing of your context except this text.', + }, + }, + required: ['task'], + additionalProperties: false, + }), + execute: async ({ task }, options) => { + const runCtx = toolRunContextOf(options) + const parentSignal = + (options as { abortSignal?: AbortSignal } | undefined)?.abortSignal ?? runCtx?.signal + const id = `${opts.name}-${(seq += 1)}` + const release = await acquire(parentSignal) + const emit = (e: AgentEvent) => runCtx?.emit(e) + // One controller per task: the parent's abort and our own deadline both + // stop the child (and its worker) the same way. + const controller = new AbortController() + const onParentAbort = () => controller.abort() + parentSignal?.addEventListener('abort', onParentAbort, { once: true }) + if (parentSignal?.aborted) controller.abort() + let timedOut = false + const timer = + opts.timeoutMs && opts.timeoutMs > 0 + ? setTimeout(() => { + timedOut = true + controller.abort() + }, opts.timeoutMs) + : undefined + emit({ type: 'subagent.start', id, name: opts.name, task }) + try { + const forward = (event: AgentEvent) => + emit({ type: 'subagent.event', id, name: opts.name, event }) + const hostTools = gatedHostTools(runCtx?.approve) + const run = opts.worker ? runInWorker : runInProcess + const out = await run(task, controller.signal, forward, hostTools).catch((err: unknown) => { + if (timedOut) + throw new Error(`subagent "${opts.name}" timed out after ${opts.timeoutMs} ms`) + throw err + }) + runCtx?.addUsage(out.usage) + emit({ type: 'subagent.complete', id, name: opts.name, text: out.text, usage: out.usage }) + return clip(out.text, outputMax) + } catch (err) { + emit({ + type: 'subagent.error', + id, + name: opts.name, + error: err instanceof Error ? err.message : String(err), + }) + throw err + } finally { + if (timer !== undefined) clearTimeout(timer) + parentSignal?.removeEventListener('abort', onParentAbort) + release() + } + }, + }) as AgentTool + t.promptHint = '{ task: string }' + if (opts.readOnly) t.readOnly = true + return t +} diff --git a/src/subagent/worker.ts b/src/subagent/worker.ts new file mode 100644 index 0000000..53629d8 --- /dev/null +++ b/src/subagent/worker.ts @@ -0,0 +1,154 @@ +import { dynamicTool, jsonSchema, type LanguageModel, type ToolSet } from 'ai' +import { createAgent } from '../agent/runner.js' +import type { BrowserAgentConfig } from '../config.js' +import type { AgentEvent } from '../events.js' +import { buildModelFromStage, resolveStage } from '../providers/registry.js' +import type { ProviderModelSpec } from '../providers/types.js' +import { MemoryCredentialStore } from '../secrets/store.js' +import { markReadOnly } from '../tools/approval.js' +import { + safePost, + type MessageEndpoint, + type ParentToWorker, + type ProxiedToolSpec, +} from './protocol.js' + +export interface ServeSubagentOptions { + /** + * Build the child's model from its spec. Pass one when your bundler can't + * resolve the core's dynamic provider imports inside a worker (Vite): import + * the provider factories statically in the worker module and build here. + * Default: the core registry (dynamic import of the optional peer). + */ + resolveModel?: (spec: ProviderModelSpec) => LanguageModel | Promise + /** + * Worker-local tools — they run in the worker thread, so CPU-heavy work + * (search, parsing, simulation) never blocks the page. + */ + tools?: ToolSet + /** The worker's message endpoint (default: the worker global scope). */ + scope?: MessageEndpoint +} + +// The key arrives with the task and lives only in this in-memory store, so +// the registry builds the model without the inline-key warning. +const SUBAGENT_REF = '__subagent__' + +const defaultResolveModel = (spec: ProviderModelSpec): Promise => { + const { apiKey, ...rest } = spec + const credentials = apiKey ? new MemoryCredentialStore({ [SUBAGENT_REF]: apiKey }) : undefined + return buildModelFromStage( + resolveStage(apiKey ? { ...rest, credentialRef: SUBAGENT_REF } : rest, undefined, 'subagent'), + credentials, + ) +} + +/** + * Serve subagent tasks inside a Web Worker. Call it once at the top of your + * worker module; the parent's `createSubagentTool({ worker })` posts a task, + * this builds a child agent (worker-local tools + the parent's host tools + * proxied over RPC), runs it, streams its events back and answers with the + * final text. + * + * ```ts + * // agent.worker.ts + * import { serveSubagentWorker } from '@dudko.dev/agent-web' + * import { createGoogleGenerativeAI } from '@ai-sdk/google' + * serveSubagentWorker({ + * resolveModel: (spec) => createGoogleGenerativeAI({ apiKey: spec.apiKey })(spec.model), + * tools: { heavy_search: defineTool({ … }) }, + * }) + * ``` + */ +export const serveSubagentWorker = (opts: ServeSubagentOptions = {}): void => { + const scope = opts.scope ?? (globalThis as unknown as MessageEndpoint) + let controller: AbortController | undefined + const pending = new Map void; reject: (e: Error) => void }>() + let callSeq = 0 + + const proxy = (spec: ProxiedToolSpec) => { + const t = dynamicTool({ + description: spec.description, + inputSchema: jsonSchema(spec.inputSchema as Parameters[0]), + execute: (input, options) => + new Promise((resolve, reject) => { + const callId = `call-${(callSeq += 1)}` + const signal = options?.abortSignal + if (signal?.aborted) return reject(new Error('aborted')) + pending.set(callId, { resolve, reject }) + signal?.addEventListener( + 'abort', + () => { + if (pending.delete(callId)) reject(new Error('aborted')) + }, + { once: true }, + ) + safePost(scope, { type: 'tool-call', callId, name: spec.name, input }) + }), + }) + return spec.readOnly ? markReadOnly(t) : t + } + + const run = async (msg: Extract): Promise => { + controller = new AbortController() + let runError: string | undefined + try { + const { model: spec, ...rest } = msg.config as { model: ProviderModelSpec } & Record< + string, + unknown + > + const model = await (opts.resolveModel ?? defaultResolveModel)(spec) + const proxied = Object.fromEntries(msg.proxied.map((p) => [p.name, proxy(p)])) + const agent = await createAgent({ + ...(rest as Partial), + model, + tools: { ...opts.tools, ...proxied }, + }) + const result = await agent.run(msg.task, { + signal: controller.signal, + onEvent: (event: AgentEvent) => { + if (event.type === 'error' && event.phase === 'run') runError = event.error + safePost(scope, { type: 'event', event }) + }, + }) + if (result.stopped) safePost(scope, { type: 'error', error: 'aborted' }) + else if (runError) safePost(scope, { type: 'error', error: runError }) + else + safePost(scope, { + type: 'done', + text: result.final, + usage: result.usage, + steps: result.steps, + }) + } catch (err) { + safePost(scope, { type: 'error', error: err instanceof Error ? err.message : String(err) }) + } + } + + scope.addEventListener('message', (ev: { data: unknown }) => { + const msg = ev.data as ParentToWorker + switch (msg?.type) { + case 'run': + void run(msg) + break + case 'abort': + controller?.abort() + for (const [id, p] of pending) { + pending.delete(id) + p.reject(new Error('aborted')) + } + break + case 'tool-result': { + const p = pending.get(msg.callId) + if (!p) break + pending.delete(msg.callId) + if (msg.ok) p.resolve(msg.output) + else + p.reject( + new Error(typeof msg.output === 'string' ? msg.output : JSON.stringify(msg.output)), + ) + break + } + } + }) +} diff --git a/src/thinking.ts b/src/thinking.ts new file mode 100644 index 0000000..17cc2eb --- /dev/null +++ b/src/thinking.ts @@ -0,0 +1,109 @@ +/** + * Thinking (a.k.a. reasoning) activation, portable across providers. + * + * The AI SDK exposes a top-level `reasoning` level on every call and translates + * it to each provider's native API (OpenAI reasoning effort, Anthropic adaptive + * thinking, Gemini thinking level, …). An exact token budget is not portable, + * so a `budgetTokens` is mapped to the provider-specific options of the two + * providers that take one (Anthropic, Google); provider options win over the + * portable level by the SDK's precedence rules. + */ + +export type ThinkingLevel = + 'provider-default' | 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' + +export interface ThinkingConfig { + /** Portable effort level (default 'medium'). */ + level?: ThinkingLevel + /** Exact thinking budget — mapped to Anthropic `thinking` / Google `thinkingConfig`. */ + budgetTokens?: number + /** Stream the model's thoughts where the provider can (default true). */ + includeThoughts?: boolean +} + +/** + * `false`/undefined → leave the provider default alone; `true` → 'medium'; + * a level string → that level; an object → full control. + */ +export type ThinkingSetting = boolean | ThinkingLevel | ThinkingConfig + +export type ThinkingStage = 'planner' | 'executor' | 'replanner' | 'synthesizer' + +/** Provider-keyed options, the shape of the AI SDK's `providerOptions`. */ +export type ProviderOptionsMap = Record> + +export interface ResolvedThinking { + reasoning?: ThinkingLevel + providerOptions?: ProviderOptionsMap +} + +const normalize = (setting: ThinkingSetting | undefined): ThinkingConfig | undefined => { + if (setting === undefined || setting === false) return undefined + if (setting === true) return { level: 'medium' } + if (typeof setting === 'string') return { level: setting } + return setting +} + +/** + * Turn a thinking setting into call options: the portable `reasoning` level + * plus the provider-specific extras for an exact budget or streamed thoughts. + * Pure — unit-tested. + */ +export const resolveThinking = (setting: ThinkingSetting | undefined): ResolvedThinking => { + const cfg = normalize(setting) + if (!cfg) return {} + const level = cfg.level ?? 'medium' + if (level === 'none') return { reasoning: 'none' } + const includeThoughts = cfg.includeThoughts !== false + const budget = cfg.budgetTokens + if (typeof budget === 'number' && budget > 0) { + return { + reasoning: level, + providerOptions: { + anthropic: { thinking: { type: 'enabled', budgetTokens: budget } }, + google: { thinkingConfig: { thinkingBudget: budget, includeThoughts } }, + }, + } + } + if (!includeThoughts) return { reasoning: level } + return { + reasoning: level, + providerOptions: { + google: { thinkingConfig: { includeThoughts: true } }, + openai: { reasoningSummary: 'auto' }, + }, + } +} + +/** The setting that applies to one stage: a stage entry wins over the top level. */ +export const thinkingFor = ( + stage: ThinkingStage, + top: ThinkingSetting | undefined, + perStage: Partial> | undefined, +): ThinkingSetting | undefined => (perStage && stage in perStage ? perStage[stage] : top) + +const isPlainObject = (v: unknown): v is Record => + typeof v === 'object' && v !== null && !Array.isArray(v) + +const deepMerge = (a: Record, b: Record) => { + const out: Record = { ...a } + for (const [k, v] of Object.entries(b)) { + out[k] = isPlainObject(out[k]) && isPlainObject(v) ? deepMerge(out[k], v) : v + } + return out +} + +/** + * Deep-merge provider-option maps left to right (later wins). Returns undefined + * when there is nothing to send, so callers can spread it unconditionally. + */ +export const mergeProviderOptions = ( + ...maps: (ProviderOptionsMap | undefined)[] +): ProviderOptionsMap | undefined => { + let out: Record | undefined + for (const m of maps) { + if (!m || Object.keys(m).length === 0) continue + out = out ? deepMerge(out, m) : { ...m } + } + return out as ProviderOptionsMap | undefined +} diff --git a/src/tools/approval.ts b/src/tools/approval.ts new file mode 100644 index 0000000..1ef7930 --- /dev/null +++ b/src/tools/approval.ts @@ -0,0 +1,255 @@ +import type { Tool } from 'ai' +import type { IPlanStep } from '../agent/loop-types.js' + +/** + * Tool consent. Every tool call passes a gate that decides allow / ask / deny: + * + * - `autopilot` — run everything (default; explicit `rules` still apply); + * - `ask-writes` — run read-only tools, ask before anything that may change state; + * - `ask-all` — ask before every call; + * - `read-only` — run read-only tools, refuse the rest without asking. + * + * A tool is read-only when its MCP annotations say so (`readOnlyHint`) or the + * host marked it (`defineTool({ readOnly: true })` / `markReadOnly`). Unknown = + * may write: the safe assumption. + */ +export type ToolApprovalMode = 'autopilot' | 'ask-writes' | 'ask-all' | 'read-only' +export type ToolPermission = 'allow' | 'ask' | 'deny' + +export const TOOL_APPROVAL_MODES: readonly ToolApprovalMode[] = [ + 'autopilot', + 'ask-writes', + 'ask-all', + 'read-only', +] + +export interface ToolApprovalRequest { + /** Unique per request — echo it back from a UI. */ + id: string + toolName: string + input: unknown + readOnly: boolean + step?: IPlanStep +} + +/** + * `true`/`false`, or an object. `remember: true` on an approval allows this + * tool for the rest of the agent's life ("always allow"). + */ +export type ToolApprovalDecision = + boolean | { approved: boolean; reason?: string; remember?: boolean } + +export interface ToolApprovalConfig { + /** Default policy (default 'autopilot'). */ + mode?: ToolApprovalMode + /** + * Per-tool overrides by exact name or `*` glob, e.g. `{ 'github__delete_*': 'deny' }`. + * They win over the mode — an explicit 'ask' still asks under autopilot. + */ + rules?: Record + /** Resolves an approval request (a UI prompt, a policy service, …). */ + onRequest?: (req: ToolApprovalRequest) => ToolApprovalDecision | Promise + /** No decision within this many ms → deny. Default: wait indefinitely. */ + timeoutMs?: number +} + +/** Raised by a denied tool call; recorded as a failed call the replanner sees. */ +export class ToolDeniedError extends Error { + constructor(reason?: string) { + super( + `Tool call denied by the user${reason ? `: ${reason}` : ''}. Do not retry it; continue without it, or report what is blocked.`, + ) + this.name = 'ToolDeniedError' + } + + toJSON(): string { + return this.message + } +} + +/** Mark a tool as read-only so `ask-writes` / `read-only` let it through. Returns it. */ +export const markReadOnly = (tool: T, readOnly = true): T => { + ;(tool as { readOnly?: boolean }).readOnly = readOnly + return tool +} + +export const isReadOnlyTool = (tool: Tool | undefined): boolean => + (tool as { readOnly?: unknown } | undefined)?.readOnly === true + +const globToRegExp = (glob: string): RegExp => + new RegExp( + `^${glob + .split('*') + .map((s) => s.replace(/[.+?^${}()|[\]\\]/g, '\\$&')) + .join('.*')}$`, + ) + +/** The most specific matching rule: an exact name, else the longest glob. Pure. */ +export const matchRule = ( + rules: Record | undefined, + toolName: string, +): ToolPermission | undefined => { + if (!rules) return undefined + if (Object.hasOwn(rules, toolName)) return rules[toolName] + let best: { len: number; perm: ToolPermission } | undefined + for (const [pattern, perm] of Object.entries(rules)) { + if (!pattern.includes('*')) continue + if (!globToRegExp(pattern).test(toolName)) continue + const len = pattern.replace(/\*/g, '').length + if (!best || len > best.len) best = { len, perm } + } + return best?.perm +} + +/** + * What the gate does for one call, before any human is involved. Order: + * remembered "always allow" → most specific rule → mode default. Pure. + */ +export const decideToolPermission = (opts: { + toolName: string + readOnly: boolean + mode: ToolApprovalMode + rules?: Record + remembered?: ReadonlySet +}): ToolPermission => { + if (opts.remembered?.has(opts.toolName)) return 'allow' + const rule = matchRule(opts.rules, opts.toolName) + if (rule) return rule + switch (opts.mode) { + case 'autopilot': + return 'allow' + case 'read-only': + return opts.readOnly ? 'allow' : 'deny' + case 'ask-writes': + return opts.readOnly ? 'allow' : 'ask' + case 'ask-all': + return 'ask' + } +} + +const normalizeDecision = ( + d: ToolApprovalDecision, +): { approved: boolean; reason?: string; remember?: boolean } => + typeof d === 'boolean' + ? { approved: d } + : { approved: Boolean(d?.approved), reason: d?.reason, remember: d?.remember } + +/** Mutable approval state shared by every run of one agent. */ +export interface ApprovalState { + mode: ToolApprovalMode + readonly config: ToolApprovalConfig + /** Tools the user chose to "always allow". */ + readonly remembered: Set +} + +export const createApprovalState = (config: ToolApprovalConfig | undefined): ApprovalState => ({ + mode: config?.mode ?? 'autopilot', + config: config ?? {}, + remembered: new Set(), +}) + +let seq = 0 + +export interface ApprovalEvents { + requested: (req: ToolApprovalRequest) => void + resolved: (r: { + id: string + name: string + approved: boolean + reason?: string + automatic: boolean + }) => void +} + +/** + * Run the gate for one call. Resolves when the call may proceed; throws + * {@link ToolDeniedError} otherwise. `alwaysAllow` short-circuits (built-ins). + */ +export const gateToolCall = async ( + state: ApprovalState, + call: { toolName: string; input: unknown; readOnly: boolean; step?: IPlanStep }, + events: ApprovalEvents, + signal?: AbortSignal, +): Promise => { + const permission = decideToolPermission({ + toolName: call.toolName, + readOnly: call.readOnly, + mode: state.mode, + rules: state.config.rules, + remembered: state.remembered, + }) + if (permission === 'allow') return + const id = `approval-${Date.now().toString(36)}-${(seq += 1)}` + if (permission === 'deny') { + const reason = + state.mode === 'read-only' && !matchRule(state.config.rules, call.toolName) + ? 'the agent is in read-only mode' + : 'blocked by policy' + events.resolved({ id, name: call.toolName, approved: false, reason, automatic: true }) + throw new ToolDeniedError(reason) + } + const onRequest = state.config.onRequest + if (!onRequest) { + const reason = 'no approval handler configured' + events.resolved({ id, name: call.toolName, approved: false, reason, automatic: true }) + throw new ToolDeniedError(reason) + } + const req: ToolApprovalRequest = { id, ...call } + events.requested(req) + let decision: { approved: boolean; reason?: string; remember?: boolean } + try { + decision = normalizeDecision( + await withDeadline(Promise.resolve(onRequest(req)), state.config.timeoutMs, signal), + ) + } catch (err) { + const reason = err instanceof Error ? err.message : String(err) + events.resolved({ id, name: call.toolName, approved: false, reason, automatic: true }) + throw new ToolDeniedError(reason) + } + if (decision.approved && decision.remember) state.remembered.add(call.toolName) + events.resolved({ + id, + name: call.toolName, + approved: decision.approved, + reason: decision.reason, + automatic: false, + }) + if (!decision.approved) throw new ToolDeniedError(decision.reason) +} + +const withDeadline = ( + p: Promise, + timeoutMs: number | undefined, + signal?: AbortSignal, +): Promise => { + if ((!timeoutMs || timeoutMs <= 0) && !signal) return p + return new Promise((resolve, reject) => { + let timer: ReturnType | undefined + const done = () => { + if (timer !== undefined) clearTimeout(timer) + signal?.removeEventListener('abort', onAbort) + } + const onAbort = () => { + done() + reject(new Error('the run was stopped while waiting for approval')) + } + if (signal?.aborted) return onAbort() + signal?.addEventListener('abort', onAbort, { once: true }) + if (timeoutMs && timeoutMs > 0) { + timer = setTimeout(() => { + done() + reject(new Error(`no approval decision within ${timeoutMs} ms`)) + }, timeoutMs) + } + p.then( + (v) => { + done() + resolve(v) + }, + (e) => { + done() + reject(e) + }, + ) + }) +} diff --git a/src/tools/define.ts b/src/tools/define.ts index 1916144..29c141e 100644 --- a/src/tools/define.ts +++ b/src/tools/define.ts @@ -14,7 +14,9 @@ import type { AgentTool } from './types.js' * * Optionally add `promptHint` — a short parameter hint like * "{ text: string, x?: number }" — which the prompted/salvage path shows to - * weak local models that can't do native function-calling. + * weak local models that can't do native function-calling — and `readOnly: + * true` for tools that never change state (the `ask-writes` and `read-only` + * approval modes let those run without asking). * * @example * const tools = { @@ -27,10 +29,12 @@ import type { AgentTool } from './types.js' * } */ export const defineTool = ( - config: Tool & { promptHint?: string }, + config: Tool & { promptHint?: string; readOnly?: boolean }, ): AgentTool => { - const { promptHint, ...rest } = config + const { promptHint, readOnly, ...rest } = config const t = tool(rest as Tool) as AgentTool if (promptHint) t.promptHint = promptHint + // Read-only tools pass the consent gate in 'ask-writes' / 'read-only' modes. + if (readOnly) t.readOnly = true return t } diff --git a/src/tools/prompted.ts b/src/tools/prompted.ts index 76b1719..963def3 100644 --- a/src/tools/prompted.ts +++ b/src/tools/prompted.ts @@ -6,15 +6,19 @@ import { promptHintOf, type AgentToolSet } from './types.js' /** * Render a ToolSet into a plain-text catalogue for the prompted/salvage path: * `- name(hint): description`. The hint comes from `defineTool({ promptHint })` - * when present; weak local models lean on it to fill args correctly. + * when present, else from `hintOf` (the agent derives one from the tool's JSON + * schema, so MCP tools get one too); weak local models lean on it to fill args. */ -export const renderCatalog = (tools: AgentToolSet): string => { +export const renderCatalog = ( + tools: AgentToolSet, + hintOf?: (name: string) => string | undefined, +): string => { const names = Object.keys(tools) if (names.length === 0) return '(no tools)' return names .map((name) => { const t = tools[name] - const hint = promptHintOf(t) ?? '' + const hint = promptHintOf(t) ?? hintOf?.(name) ?? '' const desc = typeof t.description === 'string' ? t.description : '' return `- ${name}(${hint}): ${desc}` }) diff --git a/src/tools/search.ts b/src/tools/search.ts new file mode 100644 index 0000000..5f359f1 --- /dev/null +++ b/src/tools/search.ts @@ -0,0 +1,175 @@ +import { jsonSchema, tool, type Tool, type ToolSet } from 'ai' +import { isReadOnlyTool } from './approval.js' + +/** + * Tool search for large catalogues. Several MCP servers easily add up to + * hundreds of tools — sending every schema on every call wastes the context + * window (and OpenAI rejects more than 128 tools outright). In 'search' mode + * the executor starts each step with a handful of tools and a `find_tools` + * meta-tool that activates more on demand. + */ +export interface ToolCatalogEntry { + name: string + description: string + /** The MCP server a tool came from ("server__tool" prefix), when known. */ + server?: string + readOnly?: boolean +} + +export const FIND_TOOLS_NAME = 'find_tools' + +/** Build catalogue entries from a ToolSet ("server__tool" names yield a server). */ +export const toolCatalogOf = (tools: ToolSet): ToolCatalogEntry[] => + Object.entries(tools).map(([name, t]) => { + const sep = name.indexOf('__') + return { + name, + description: typeof t.description === 'string' ? t.description : '', + ...(sep > 0 ? { server: name.slice(0, sep) } : {}), + readOnly: isReadOnlyTool(t as Tool), + } + }) + +/** Lowercased word tokens; snake_case, kebab-case and camelCase are split. */ +export const tokenize = (s: string): string[] => + (s ?? '') + .replace(/([a-z0-9])([A-Z])/g, '$1 $2') + .toLowerCase() + .split(/[^a-z0-9]+/) + .filter(Boolean) + +const hits = (term: string, words: string[]): boolean => + words.some( + (w) => + w === term || + (term.length >= 3 && (w.startsWith(term) || (term.startsWith(w) && w.length >= 3))), + ) + +export interface SearchToolsOptions { + /** Only tools from this server. */ + server?: string + /** Max results (default 8, capped at 20). */ + limit?: number + /** Names to leave out (e.g. already active). */ + exclude?: ReadonlySet +} + +/** + * Rank catalogue entries against a keyword query: each query term scores + * 3 for a name hit, 2 for a server hit and 1 for a description hit (prefix + * matching for terms of 3+ chars). Ties keep catalogue order; zero scores are + * dropped. Pure and deterministic — no embeddings, no model call. + */ +export const searchTools = ( + catalog: ToolCatalogEntry[], + query: string, + opts: SearchToolsOptions = {}, +): ToolCatalogEntry[] => { + const terms = [...new Set(tokenize(query))] + const limit = Math.min(Math.max(1, opts.limit ?? 8), 20) + const server = opts.server?.toLowerCase() + const scored: { entry: ToolCatalogEntry; score: number; index: number }[] = [] + catalog.forEach((entry, index) => { + if (opts.exclude?.has(entry.name)) return + if (server && entry.server?.toLowerCase() !== server) return + const nameWords = tokenize( + entry.server ? entry.name.slice(entry.server.length + 2) : entry.name, + ) + const serverWords = tokenize(entry.server ?? '') + const descWords = tokenize(entry.description) + let score = 0 + for (const term of terms) { + if (hits(term, nameWords)) score += 3 + if (hits(term, serverWords)) score += 2 + if (hits(term, descWords)) score += 1 + } + // A server filter with an empty query lists that server's tools. + if (terms.length === 0 && server) score = 1 + if (score > 0) scored.push({ entry, score, index }) + }) + scored.sort((a, b) => b.score - a.score || a.index - b.index) + return scored.slice(0, limit).map((s) => s.entry) +} + +/** + * The planner's view of a large catalogue: grouped by server, short + * descriptions, within a character budget. + */ +export const renderSearchCatalog = (catalog: ToolCatalogEntry[], budget = 12_000): string => { + if (catalog.length === 0) return '(no tools)' + const groups = new Map() + for (const e of catalog) { + const key = e.server ?? '(host)' + groups.set(key, [...(groups.get(key) ?? []), e]) + } + const lines: string[] = [] + let used = 0 + let shown = 0 + for (const [server, entries] of groups) { + const head = `[${server}] ${entries.length} tool(s)` + if (used + head.length > budget) break + lines.push(head) + used += head.length + 1 + for (const e of entries) { + const desc = e.description.replace(/\s+/g, ' ').trim().slice(0, 60) + const line = `- ${e.name}${desc ? `: ${desc}` : ''}` + if (used + line.length > budget) break + lines.push(line) + used += line.length + 1 + shown += 1 + } + } + const rest = catalog.length - shown + if (rest > 0) + lines.push(`… ${rest} more tool(s) — the executor can find them with ${FIND_TOOLS_NAME}`) + return lines.join('\n') +} + +/** + * The `find_tools` meta-tool. `onFound` receives the matched names so the + * caller can activate them (native: prepareStep activeTools; prompted: the + * next round's catalogue). + */ +export const createFindToolsTool = ( + catalog: () => ToolCatalogEntry[], + onFound: (query: string, names: string[]) => void, + hintOf?: (name: string) => string | undefined, +): Tool => { + const t = tool({ + description: + 'Search the full tool catalogue by keywords and make the matching tools callable from the next step on. ' + + 'Use it whenever the tool you need is not in your current tool list.', + inputSchema: jsonSchema<{ query: string; server?: string; limit?: number }>({ + type: 'object', + properties: { + query: { type: 'string', description: 'Keywords: what the tool should do' }, + server: { type: 'string', description: 'Optional: only tools of this MCP server' }, + limit: { type: 'number', description: 'Max results (default 8, max 20)' }, + }, + required: ['query'], + additionalProperties: false, + }), + execute: async ({ query, server, limit }) => { + const found = searchTools(catalog(), query, { server, limit }) + onFound( + query, + found.map((f) => f.name), + ) + return { + tools: found.map((f) => ({ + name: f.name, + description: f.description, + ...(f.server ? { server: f.server } : {}), + readOnly: Boolean(f.readOnly), + ...(hintOf?.(f.name) ? { args: hintOf(f.name) } : {}), + })), + note: found.length + ? 'These tools are now callable.' + : 'No tool matched; try other keywords or a server name.', + } + }, + }) + ;(t as { readOnly?: boolean }).readOnly = true + ;(t as { promptHint?: string }).promptHint = '{ query: string, server?: string, limit?: number }' + return t +} diff --git a/src/tools/types.ts b/src/tools/types.ts index 17c4764..b9a5232 100644 --- a/src/tools/types.ts +++ b/src/tools/types.ts @@ -8,7 +8,7 @@ import type { Tool, ToolSet } from 'ai' * by the prompted/salvage path when it renders the tool catalogue into a text * prompt for weak local models. */ -export type AgentTool = Tool & { promptHint?: string } +export type AgentTool = Tool & { promptHint?: string; readOnly?: boolean } /** A named collection of tools — a plain AI SDK ToolSet, usable directly. */ export type AgentToolSet = ToolSet diff --git a/src/tools/wrap.ts b/src/tools/wrap.ts new file mode 100644 index 0000000..efa2eab --- /dev/null +++ b/src/tools/wrap.ts @@ -0,0 +1,223 @@ +import { asSchema, type Tool, type ToolSet } from 'ai' +import type { IPlanStep, IUsage } from '../agent/loop-types.js' +import type { AgentEvent } from '../events.js' +import { ToolBudgetError } from '../limits.js' +import { + decideToolPermission, + gateToolCall, + isReadOnlyTool, + type ApprovalState, +} from './approval.js' +import { promptHintOf } from './types.js' + +/** + * What a tool's `execute` can reach of the run that called it, passed as + * `options.agentRun`. Subagent tools use it to stream nested events and to + * charge their usage to the parent run. + */ +export interface ToolRunContext { + emit: (event: AgentEvent) => void + /** The plan step being executed, when there is one. */ + step?: IPlanStep + signal?: AbortSignal + /** Add tokens spent on the parent's behalf (e.g. by a subagent) to the run. */ + addUsage: (usage: IUsage) => void + /** + * Run the agent's consent gate for a call made on its behalf (e.g. a host + * tool a subagent calls). Resolves when allowed, throws ToolDeniedError. + */ + approve: (toolName: string, input: unknown, readOnly: boolean) => Promise +} + +/** Read the run context the agent passes to a tool's execute (undefined outside a run). */ +export const toolRunContextOf = (options: unknown): ToolRunContext | undefined => + (options as { agentRun?: ToolRunContext } | undefined)?.agentRun + +const truncated = (s: string, max: number): string => + s.length > max ? `${s.slice(0, max)}… [truncated ${s.length - max} chars]` : s + +const toText = (v: unknown): string => { + if (typeof v === 'string') return v + try { + return JSON.stringify(v) ?? String(v) + } catch { + return String(v) + } +} + +type ToModelOutput = NonNullable + +/** + * Cap what the MODEL sees of a tool result at `maxChars` (the raw output still + * reaches events and the trace). A tool's own `toModelOutput` runs first. + */ +export const limitModelOutput = ( + original: Tool['toModelOutput'], + maxChars: number, +): ToModelOutput => + (async (opts: Parameters[0]) => { + const base = original + ? await original(opts) + : typeof opts.output === 'string' + ? ({ type: 'text', value: opts.output } as const) + : ({ type: 'json', value: (opts.output ?? null) as never } as const) + if (base.type === 'text' || base.type === 'error-text') { + return { ...base, value: truncated(base.value, maxChars) } + } + if (base.type === 'json' || base.type === 'error-json') { + const text = toText(base.value) + if (text.length <= maxChars) return base + return { + type: base.type === 'json' ? 'text' : 'error-text', + value: truncated(text, maxChars), + } + } + return base + }) as ToModelOutput + +export interface WrapToolsOptions { + approval: ApprovalState + /** Names that skip the gate and the call budget (built-in meta-tools). */ + builtins: ReadonlySet + emit: (event: AgentEvent) => void + step: () => IPlanStep | undefined + signal?: AbortSignal + addUsage: (usage: IUsage) => void + /** Called before each counted call; throw to refuse it. */ + countCall: () => void + maxToolOutputChars: number +} + +/** + * Wrap every tool for one run: the consent gate, the tool-call budget, the + * model-facing output cap, and the run context passed as `options.agentRun`. + * When the gate allows a call synchronously, `execute` is invoked without an + * extra await, so a streaming tool keeps returning its iterable. + */ +export const wrapToolsForRun = (tools: ToolSet, o: WrapToolsOptions): ToolSet => { + const out: ToolSet = {} + const gate = (toolName: string, input: unknown, readOnly: boolean, signal?: AbortSignal) => + gateToolCall( + o.approval, + { toolName, input, readOnly, step: o.step() }, + { + requested: (req) => + o.emit({ + type: 'tool.approval-requested', + id: req.id, + name: req.toolName, + input: req.input, + readOnly: req.readOnly, + step: req.step, + }), + resolved: (r) => o.emit({ type: 'tool.approval-resolved', ...r }), + }, + signal ?? o.signal, + ) + for (const [name, t] of Object.entries(tools)) { + const execute = (t as { execute?: (input: unknown, options: unknown) => unknown }).execute + const builtin = o.builtins.has(name) + const readOnly = builtin || isReadOnlyTool(t) + const wrapped: Record = { ...t } + if (o.maxToolOutputChars > 0) { + wrapped.toModelOutput = limitModelOutput(t.toModelOutput, o.maxToolOutputChars) + } + if (typeof execute === 'function') { + wrapped.execute = (input: unknown, options: unknown) => { + const run = (): unknown => { + if (!builtin) o.countCall() + const agentRun: ToolRunContext = { + emit: o.emit, + step: o.step(), + signal: o.signal, + addUsage: o.addUsage, + approve: (toolName, toolInput, ro) => gate(toolName, toolInput, ro), + } + return execute.call(t, input, { ...(options as object), agentRun }) + } + if (builtin) return run() + const permission = decideToolPermission({ + toolName: name, + readOnly, + mode: o.approval.mode, + rules: o.approval.config.rules, + remembered: o.approval.remembered, + }) + if (permission === 'allow') return run() + return gate( + name, + input, + readOnly, + (options as { abortSignal?: AbortSignal } | undefined)?.abortSignal, + ).then(run) + } + } + out[name] = wrapped as Tool + } + return out +} + +/** Throws ToolBudgetError once `cap` calls were made (cap ≤ 0 = unlimited). */ +export const createCallCounter = (cap: number | undefined) => { + let calls = 0 + return { + count: (): void => { + if (cap && cap > 0 && calls >= cap) throw new ToolBudgetError(cap) + calls += 1 + }, + get calls() { + return calls + }, + get exhausted() { + return Boolean(cap && cap > 0 && calls >= cap) + }, + } +} + +type JsonSchemaLike = { + type?: string | string[] + properties?: Record + required?: string[] + items?: JsonSchemaLike + enum?: unknown[] +} + +const typeHint = (s: JsonSchemaLike | undefined, depth: number): string => { + if (!s) return 'any' + if (Array.isArray(s.enum) && s.enum.length > 0 && s.enum.length <= 8) { + return s.enum.map((v) => JSON.stringify(v)).join(' | ') + } + const type = Array.isArray(s.type) ? s.type.find((x) => x !== 'null') : s.type + if (type === 'array') return `${typeHint(s.items, depth)}[]` + if (type === 'object' || s.properties) { + if (depth > 1 || !s.properties) return 'object' + return objectHint(s, depth + 1) + } + return type ?? 'any' +} + +const objectHint = (s: JsonSchemaLike, depth = 0): string => { + const required = new Set(s.required ?? []) + const props = Object.entries(s.properties ?? {}) + if (props.length === 0) return '{}' + return `{ ${props + .slice(0, 12) + .map(([k, v]) => `${k}${required.has(k) ? '' : '?'}: ${typeHint(v, depth)}`) + .join(', ')}${props.length > 12 ? ', …' : ''} }` +} + +/** + * A short parameter hint ("{ text: string, x?: number }") derived from a + * tool's JSON schema — so tools without a hand-written `promptHint` (every MCP + * tool) still tell a prompted model what to pass. Undefined when unknown. + */ +export const schemaHintOf = async (t: Tool): Promise => { + const own = promptHintOf(t) + if (own) return own + try { + const schema = (await asSchema(t.inputSchema as never).jsonSchema) as JsonSchemaLike + return schema && typeof schema === 'object' ? objectHint(schema) : undefined + } catch { + return undefined + } +} diff --git a/tests/features.test.ts b/tests/features.test.ts new file mode 100644 index 0000000..7490cad --- /dev/null +++ b/tests/features.test.ts @@ -0,0 +1,733 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { z } from 'zod' +import { + checkLimits, + createAgent, + createSubagentTool, + decideToolPermission, + defaultPrompts, + defineSkill, + defineTool, + limitModelOutput, + matchRule, + MemoryStore, + parseSkillMarkdown, + renderSearchCatalog, + resolveThinking, + searchTools, + serveSubagentWorker, + thinkingFor, + type AgentEvent, +} from '../dist/index.js' +import { scriptedModel, stageOf, type CallInfo, type Reply } from './helpers/scripted-model.ts' + +const plan = (steps: string[], extra: Record = {}) => + JSON.stringify({ + thought: 'plan', + steps: steps.map((description) => ({ description })), + ...extra, + }) + +const collect = () => { + const events: AgentEvent[] = [] + return { + events, + onEvent: (e: AgentEvent) => events.push(e), + types: () => events.map((e) => e.type), + } +} + +// ── pure helpers ──────────────────────────────────────────────────────────── + +test('resolveThinking maps settings to a reasoning level and provider options', () => { + assert.deepEqual(resolveThinking(undefined), {}) + assert.deepEqual(resolveThinking(false), {}) + assert.deepEqual(resolveThinking('none'), { reasoning: 'none' }) + const on = resolveThinking(true) + assert.equal(on.reasoning, 'medium') + assert.deepEqual(on.providerOptions?.google, { thinkingConfig: { includeThoughts: true } }) + assert.deepEqual(on.providerOptions?.openai, { reasoningSummary: 'auto' }) + const budget = resolveThinking({ level: 'high', budgetTokens: 2048 }) + assert.equal(budget.reasoning, 'high') + assert.deepEqual(budget.providerOptions?.anthropic, { + thinking: { type: 'enabled', budgetTokens: 2048 }, + }) + assert.deepEqual(budget.providerOptions?.google, { + thinkingConfig: { thinkingBudget: 2048, includeThoughts: true }, + }) + assert.deepEqual(resolveThinking({ level: 'low', includeThoughts: false }), { reasoning: 'low' }) +}) + +test('thinkingFor: a stage entry wins over the top-level setting', () => { + assert.equal(thinkingFor('executor', 'high', { executor: 'low' }), 'low') + assert.equal(thinkingFor('planner', 'high', { executor: 'low' }), 'high') + assert.equal(thinkingFor('synthesizer', 'high', { synthesizer: false }), false) +}) + +test('checkLimits reports the broadest cap reached first', () => { + const usage = { + inputTokens: 100, + outputTokens: 50, + totalTokens: 150, + reasoningTokens: 40, + } + assert.equal(checkLimits(usage, undefined), undefined) + assert.equal(checkLimits(usage, { maxTotalTokens: 1000 }), undefined) + assert.deepEqual(checkLimits(usage, { maxTotalTokens: 150, maxInputTokens: 10 }), { + kind: 'total', + tokens: 150, + cap: 150, + }) + assert.deepEqual(checkLimits(usage, { maxInputTokens: 100 }), { + kind: 'input', + tokens: 100, + cap: 100, + }) + assert.deepEqual(checkLimits(usage, { maxOutputTokens: 20 }), { + kind: 'output', + tokens: 50, + cap: 20, + }) + assert.deepEqual(checkLimits(usage, { maxReasoningTokens: 40 }), { + kind: 'reasoning', + tokens: 40, + cap: 40, + }) +}) + +test('searchTools ranks name over server over description and honours filters', () => { + const catalog = [ + { name: 'docs__search_pages', description: 'Full-text search over the wiki', server: 'docs' }, + { name: 'github__list_issues', description: 'List issues of a repository', server: 'github' }, + { name: 'github__create_issue', description: 'Open a new issue', server: 'github' }, + { name: 'jira__find', description: 'Search jira issues by JQL', server: 'jira' }, + ] + const hits = searchTools(catalog, 'issues') + assert.deepEqual( + hits.map((h) => h.name), + ['github__list_issues', 'github__create_issue', 'jira__find'], + ) + assert.deepEqual( + searchTools(catalog, 'issue', { server: 'jira' }).map((h) => h.name), + ['jira__find'], + ) + assert.equal(searchTools(catalog, 'createIssue')[0].name, 'github__create_issue') + assert.deepEqual(searchTools(catalog, 'unrelated words'), []) + assert.equal(searchTools(catalog, 'issue', { limit: 1 }).length, 1) +}) + +test('renderSearchCatalog groups by server and stays within its budget', () => { + const catalog = Array.from({ length: 300 }, (_, i) => ({ + name: `s${i % 3}__tool_${i}`, + description: 'A tool that does a thing '.repeat(5), + server: `s${i % 3}`, + })) + const out = renderSearchCatalog(catalog, 2_000) + assert.ok(out.length < 2_200) + assert.match(out, /^\[s0\] 100 tool\(s\)/) + assert.match(out, /more tool\(s\) — the executor can find them with find_tools/) +}) + +test('parseSkillMarkdown reads frontmatter (quoted and folded values) and the body', () => { + const skill = parseSkillMarkdown( + `---\nname: release-notes\ndescription: >\n Write release notes\n from merged PRs.\nauthor: 'Jane'\n---\n# Steps\n1. List PRs\n`, + [{ path: 'template.md', content: '## Notes' }], + ) + assert.equal(skill.name, 'release-notes') + assert.equal(skill.description, 'Write release notes from merged PRs.') + assert.equal(skill.content, '# Steps\n1. List PRs') + assert.equal(skill.files?.[0].path, 'template.md') + assert.equal( + parseSkillMarkdown('---\nname: "quoted-name"\ndescription: "a: b"\n---\nx').description, + 'a: b', + ) + assert.throws(() => parseSkillMarkdown('no frontmatter'), /frontmatter/) + assert.throws(() => defineSkill({ name: 'Bad Name', description: 'd', content: '' }), /invalid/) + assert.throws(() => defineSkill({ name: 'ok', description: ' ', content: '' }), /description/) +}) + +test('tool permission: remembered > most specific rule > mode default', () => { + const rules = { 'github__*': 'ask', 'github__delete_*': 'deny', github__read: 'allow' } as const + assert.equal(matchRule(rules, 'github__delete_repo'), 'deny') + assert.equal(matchRule(rules, 'github__read'), 'allow') + assert.equal(matchRule(rules, 'github__write'), 'ask') + assert.equal(matchRule(rules, 'jira__x'), undefined) + const decide = (mode: 'autopilot' | 'ask-writes' | 'ask-all' | 'read-only', readOnly: boolean) => + decideToolPermission({ toolName: 'x', readOnly, mode }) + assert.equal(decide('autopilot', false), 'allow') + assert.equal(decide('ask-writes', false), 'ask') + assert.equal(decide('ask-writes', true), 'allow') + assert.equal(decide('ask-all', true), 'ask') + assert.equal(decide('read-only', false), 'deny') + assert.equal(decide('read-only', true), 'allow') + // An explicit rule beats autopilot; a remembered approval beats the rule. + assert.equal( + decideToolPermission({ toolName: 'github__x', readOnly: false, mode: 'autopilot', rules }), + 'ask', + ) + assert.equal( + decideToolPermission({ + toolName: 'github__x', + readOnly: false, + mode: 'ask-all', + rules, + remembered: new Set(['github__x']), + }), + 'allow', + ) +}) + +test('limitModelOutput truncates what the model sees, text and json alike', async () => { + const cap = limitModelOutput(undefined, 10) + const text = await cap({ toolCallId: 'c', input: {}, output: 'x'.repeat(50) } as never) + assert.equal(text.type, 'text') + assert.match((text as { value: string }).value, /^x{10}… \[truncated 40 chars\]$/) + const json = await cap({ toolCallId: 'c', input: {}, output: { a: 'y'.repeat(50) } } as never) + assert.equal(json.type, 'text') + const small = await cap({ toolCallId: 'c', input: {}, output: { a: 1 } } as never) + assert.deepEqual(small, { type: 'json', value: { a: 1 } }) +}) + +test('prompts are autonomous: no asking the user, consent is the system’s job', () => { + const planner = defaultPrompts.planner({ goal: 'g', toolCatalog: '-', mode: 'native' }) + assert.match(planner.system, /Never plan a step that asks the user/) + const executor = defaultPrompts.executor({ + goal: 'g', + step: 's', + index: 1, + total: 1, + toolCatalog: '-', + done: [], + mode: 'native', + }) + assert.match(executor.system, /never ask the user questions or for confirmation/) + // The goal and progress are dynamic and stay OUT of the cacheable system prompt. + assert.doesNotMatch(executor.system, /GOAL:/) + assert.match(planner.system, /TOOLS:/) +}) + +// ── native runs over a scripted model ────────────────────────────────────── + +const noteTools = (log: string[]) => ({ + add_note: defineTool({ + description: 'Add a note', + inputSchema: z.object({ text: z.string() }), + execute: async ({ text }: { text: string }) => { + log.push(`add:${text}`) + return { id: log.length } + }, + }), + list_notes: defineTool({ + description: 'List notes', + inputSchema: z.object({}), + readOnly: true, + execute: async () => { + log.push('list') + return { notes: [] } + }, + }), +}) + +/** Executor: call `calls` on the first round, then report `done`. */ +const executorReply = (info: CallInfo, calls: Reply['toolCalls'], done = 'Done.'): Reply => + info.toolResults === 0 && calls?.length ? { toolCalls: calls } : { text: done } + +test('thinking: the level and provider options reach the model, thoughts stream, usage counts them', async () => { + const { model, calls } = scriptedModel((info) => { + switch (stageOf(info)) { + case 'planner': + return { text: plan(['Add a note']) } + case 'executor': + return { + ...executorReply(info, [{ name: 'add_note', args: { text: 'hi' } }]), + reasoning: info.toolResults === 0 ? 'thinking about notes' : undefined, + usage: { reasoning: 3, cacheRead: 4 }, + } + default: + return { text: 'Added the note.', reasoning: 'summing up' } + } + }) + const log: string[] = [] + const agent = await createAgent({ + model, + tools: noteTools(log), + thinking: { level: 'high', budgetTokens: 1024 }, + stageThinking: { planner: false }, + }) + const { events, onEvent } = collect() + const result = await agent.run('add a note hi', { onEvent }) + + assert.deepEqual(log, ['add:hi']) + const executorCall = calls.find((c) => stageOf(c) === 'executor')! + assert.equal(executorCall.options.reasoning, 'high') + assert.deepEqual((executorCall.options.providerOptions as Record).anthropic, { + thinking: { type: 'enabled', budgetTokens: 1024 }, + }) + const plannerCall = calls.find((c) => stageOf(c) === 'planner')! + assert.equal(plannerCall.options.reasoning, undefined) + assert.ok( + events.some((e) => e.type === 'step.reasoning-delta' && e.delta === 'thinking about notes'), + ) + assert.ok(events.some((e) => e.type === 'final.reasoning-delta' && e.delta === 'summing up')) + assert.ok((result.usage.reasoningTokens ?? 0) >= 6) + assert.ok((result.usage.cachedInputTokens ?? 0) >= 8) + assert.equal(result.final, 'Added the note.') +}) + +test('prompt caching: a cacheable system message + an OpenAI cache key; off → plain string', async () => { + const route = (info: CallInfo): Reply => + stageOf(info) === 'planner' ? { text: plan(['Do it']) } : { text: 'ok' } + const on = scriptedModel(route) + await (await createAgent({ model: on.model, clientName: 'app' })).run('do it now please') + const prompt = on.calls[0].options.prompt as { role: string; providerOptions?: unknown }[] + assert.equal(prompt[0].role, 'system') + assert.deepEqual(prompt[0].providerOptions, { + anthropic: { cacheControl: { type: 'ephemeral' } }, + }) + assert.deepEqual(on.calls[0].options.providerOptions, { + openai: { promptCacheKey: 'app:planner' }, + }) + + const off = scriptedModel(route) + await (await createAgent({ model: off.model, promptCaching: false })).run('do it now please') + const plain = off.calls[0].options.prompt as { role: string; providerOptions?: unknown }[] + assert.equal(plain[0].providerOptions, undefined) + assert.equal(off.calls[0].options.providerOptions, undefined) +}) + +test('approval ask-writes: a write tool asks and a denial blocks it; read-only tools run freely', async () => { + const { model } = scriptedModel((info) => { + switch (stageOf(info)) { + case 'planner': + return { text: plan(['List then add']) } + case 'executor': + return executorReply(info, [ + { name: 'list_notes', args: {} }, + { name: 'add_note', args: { text: 'x' } }, + ]) + case 'replanner': + return { text: JSON.stringify({ decision: 'finish', reason: 'denied' }) } + default: + return { text: 'Could not add the note: the user said no.' } + } + }) + const log: string[] = [] + const requests: string[] = [] + const agent = await createAgent({ + model, + tools: noteTools(log), + toolApproval: { + mode: 'ask-writes', + onRequest: (req) => { + requests.push(req.toolName) + return { approved: false, reason: 'not now' } + }, + }, + }) + const { events, onEvent } = collect() + const result = await agent.run('list and add a note', { onEvent }) + assert.deepEqual(log, ['list']) + assert.deepEqual(requests, ['add_note']) + const failed = result.trace[0].toolCalls.find((c) => c.name === 'add_note') + assert.equal(failed?.ok, false) + assert.match(String(failed?.output), /denied by the user: not now/) + assert.ok(events.some((e) => e.type === 'tool.approval-requested' && e.name === 'add_note')) + assert.ok( + events.some( + (e) => e.type === 'tool.approval-resolved' && e.approved === false && e.automatic === false, + ), + ) +}) + +test('approval: "always allow" is remembered and setToolApprovalMode switches policy live', async () => { + const { model } = scriptedModel((info) => { + switch (stageOf(info)) { + case 'planner': + return { text: plan(['Add']) } + case 'executor': + return executorReply(info, [{ name: 'add_note', args: { text: 'a' } }]) + default: + return { text: 'ok' } + } + }) + const log: string[] = [] + let asked = 0 + const agent = await createAgent({ + model, + tools: noteTools(log), + toolApproval: { + mode: 'ask-all', + onRequest: () => { + asked += 1 + return { approved: true, remember: true } + }, + }, + }) + await agent.run('add a note please') + await agent.run('add a note please') + assert.equal(asked, 1) + assert.equal(log.length, 2) + + agent.setToolApprovalMode('read-only') + assert.equal(agent.toolApprovalMode, 'read-only') + // Remembered approvals still win over the mode. + await agent.run('add a note please') + assert.equal(log.length, 3) + assert.throws(() => agent.setToolApprovalMode('nope' as never), /unknown tool approval mode/) +}) + +test('approval read-only: writes are refused without asking anyone', async () => { + const { model } = scriptedModel((info) => + stageOf(info) === 'planner' + ? { text: plan(['Add']) } + : stageOf(info) === 'executor' + ? executorReply(info, [{ name: 'add_note', args: { text: 'a' } }]) + : { text: JSON.stringify({ decision: 'finish', reason: 'blocked' }) }, + ) + const log: string[] = [] + const agent = await createAgent({ + model, + tools: noteTools(log), + toolApproval: { mode: 'read-only', onRequest: () => assert.fail('must not ask') }, + }) + const { events, onEvent } = collect() + await agent.run('add a note please', { onEvent }) + assert.deepEqual(log, []) + const resolved = events.find((e) => e.type === 'tool.approval-resolved') + assert.equal(resolved?.type === 'tool.approval-resolved' && resolved.automatic, true) +}) + +test('token limits: a crossed cap stops executing, emits budget.exceeded, and still answers', async () => { + const { model } = scriptedModel((info) => { + switch (stageOf(info)) { + case 'planner': + return { text: plan(['one', 'two', 'three']), usage: { input: 40, output: 10 } } + case 'executor': + return { text: 'step done', usage: { input: 40, output: 10 } } + default: + return { text: 'Partial answer.' } + } + }) + const agent = await createAgent({ model, limits: { maxTotalTokens: 90 } }) + const { events, onEvent } = collect() + const result = await agent.run('do three things', { onEvent }) + assert.equal(result.trace.length, 1) + assert.deepEqual(result.budgetExceeded, { kind: 'total', tokens: 100, cap: 90 }) + assert.ok(events.some((e) => e.type === 'budget.exceeded')) + assert.equal(result.final, 'Partial answer.') +}) + +test('maxToolCalls: calls beyond the budget fail with a clear error', async () => { + const { model } = scriptedModel((info) => + stageOf(info) === 'planner' + ? { text: plan(['Add two']) } + : stageOf(info) === 'executor' + ? executorReply(info, [ + { name: 'add_note', args: { text: 'a' } }, + { name: 'add_note', args: { text: 'b' } }, + ]) + : { text: 'ok' }, + ) + const log: string[] = [] + const agent = await createAgent({ model, tools: noteTools(log), maxToolCalls: 1, replan: false }) + const result = await agent.run('add two notes') + assert.equal(log.length, 1) + const failed = result.trace[0].toolCalls.find((c) => !c.ok) + assert.match(String(failed?.output), /Tool-call budget exhausted/) +}) + +test('skills: a planner-picked skill is activated; load_skill activates another on demand', async () => { + const { model, calls } = scriptedModel((info) => { + switch (stageOf(info)) { + case 'planner': + return { text: plan(['Write it'], { skills: ['tone', 'ghost'] }) } + case 'executor': + return executorReply(info, [{ name: 'load_skill', args: { name: 'format' } }]) + default: + return { text: 'Written.' } + } + }) + const agent = await createAgent({ + model, + skills: [ + { name: 'tone', description: 'Friendly tone', content: 'ALWAYS-BE-FRIENDLY' }, + { name: 'format', description: 'Output format', content: 'USE-BULLETS' }, + ], + }) + const { events, onEvent } = collect() + const result = await agent.run('write a friendly note', { onEvent }) + const activated = events.filter((e) => e.type === 'skill.activated') + assert.deepEqual( + activated.map((e) => (e.type === 'skill.activated' ? `${e.name}:${e.by}` : '')), + ['tone:plan', 'format:tool'], + ) + assert.deepEqual(result.skills, ['tone', 'format']) + const planner = calls.find((c) => stageOf(c) === 'planner')! + assert.match(planner.system, /SKILLS:\n- tone: Friendly tone/) + const executor = calls.find((c) => stageOf(c) === 'executor')! + assert.match(executor.system, /ALWAYS-BE-FRIENDLY/) + const synth = calls.find((c) => stageOf(c) === 'synthesizer')! + assert.match(synth.system, /USE-BULLETS/) +}) + +test('tool search: a large catalogue starts small and find_tools activates the tool next step', async () => { + const hit: string[] = [] + const tools = Object.fromEntries( + Array.from({ length: 60 }, (_, i) => [ + `srv${i % 3}__tool_${i}`, + defineTool({ + description: i === 42 ? 'Send an invoice to a customer' : `Generic operation number ${i}`, + inputSchema: z.object({}), + execute: async () => { + hit.push(`tool_${i}`) + return 'sent' + }, + }), + ]), + ) + const offered: string[][] = [] + const { model, calls } = scriptedModel((info) => { + switch (stageOf(info)) { + case 'planner': + return { text: plan(['Send the invoice']) } + case 'executor': + offered.push(info.tools) + if (info.toolResults === 0) + return { toolCalls: [{ name: 'find_tools', args: { query: 'invoice' } }] } + if (info.toolResults === 1) return { toolCalls: [{ name: 'srv0__tool_42', args: {} }] } + return { text: 'Invoice sent.' } + default: + return { text: 'Sent.' } + } + }) + const agent = await createAgent({ model, tools }) + assert.equal(agent.toolStrategy, 'search') + const { events, onEvent } = collect() + await agent.run('send the invoice', { onEvent }) + assert.deepEqual(hit, ['tool_42']) + assert.deepEqual(offered[0], ['find_tools']) + assert.ok(offered[1].includes('srv0__tool_42')) + const discovered = events.find((e) => e.type === 'tools.discovered') + assert.ok(discovered?.type === 'tools.discovered' && discovered.names.includes('srv0__tool_42')) + const planner = calls.find((c) => stageOf(c) === 'planner')! + assert.match(planner.system, /\[srv0\] 20 tool\(s\)/) +}) + +test('tool search (prompted): discovered tools are listed and dispatchable in the next round', async () => { + const hit: string[] = [] + const tools = Object.fromEntries( + Array.from({ length: 45 }, (_, i) => [ + `t${i}`, + defineTool({ + description: i === 7 ? 'Book a meeting room' : `Other ${i}`, + inputSchema: z.object({ room: z.string().optional() }), + execute: async () => { + hit.push(`t${i}`) + return 'booked' + }, + }), + ]), + ) + const executorPrompts: string[] = [] + const { model } = scriptedModel((info) => { + const stage = stageOf(info) + if (stage === 'planner') return { text: JSON.stringify({ reply: 'ok', plan: ['Book'] }) } + if (stage === 'executor') { + const last = JSON.stringify(info.options.prompt) + executorPrompts.push(last) + const round = (info.options.prompt as { role: string }[]).filter( + (m) => m.role === 'user', + ).length + if (round === 1) { + return { + text: JSON.stringify({ + reply: 'searching', + actions: [{ tool: 'find_tools', args: { query: 'meeting room' } }], + }), + } + } + if (round === 2) { + return { + text: JSON.stringify({ + reply: 'booking', + actions: [{ tool: 't7', args: { room: 'A' } }], + }), + } + } + return { text: JSON.stringify({ reply: 'Booked room A.', actions: [] }) } + } + return { text: 'Booked.' } + }) + const agent = await createAgent({ model, tools, toolMode: 'prompted' }) + await agent.run('book a meeting room') + assert.deepEqual(hit, ['t7']) + assert.match(executorPrompts[1], /TOOLS NOW AVAILABLE/) + assert.match(executorPrompts[1], /t7\(\{ room\?: string \}\)/) +}) + +test('compaction: agent.compact summarises the stored transcript', async () => { + const { model } = scriptedModel(() => ({ text: 'SUMMARY-OF-EVERYTHING' })) + const memory = new MemoryStore() + for (let i = 0; i < 10; i += 1) { + await memory.append('s', { + role: i % 2 ? 'assistant' : 'user', + content: `message ${i} `.repeat(50), + }) + } + const agent = await createAgent({ + model, + memory, + sessionId: 's', + compaction: { keepRecentTurns: 2 }, + }) + const out = await agent.compact() + assert.equal(out.compacted, true) + assert.ok(out.afterTokens < out.beforeTokens) + const after = await memory.load('s') + assert.equal(after.length, 3) + assert.match(after[0].content, /SUMMARY-OF-EVERYTHING/) + assert.equal((await (await createAgent({ model })).compact()).compacted, false) +}) + +test('compaction: a long step log is folded into a summary between steps', async () => { + const { model, calls } = scriptedModel((info) => { + const stage = stageOf(info) + if (stage === 'planner') return { text: plan(['a', 'b', 'c', 'd']) } + if (stage === 'executor') return { text: `result ${'z'.repeat(400)}` } + if (stage === 'synthesizer') return { text: 'All done.' } + return { text: 'STEPS-SUMMARY' } + }) + const agent = await createAgent({ + model, + compaction: { thresholdTokens: 150, keepRecentSteps: 1 }, + }) + const { events, onEvent } = collect() + await agent.run('do four things', { onEvent }) + assert.ok(events.some((e) => e.type === 'context.compacted' && e.scope === 'trace')) + assert.ok(events.some((e) => e.type === 'usage' && e.phase === 'compact')) + const synth = calls.find((c) => stageOf(c) === 'synthesizer')! + assert.match(synth.user, /Summary of \d+ earlier step\(s\): STEPS-SUMMARY/) +}) + +test('step results flow into later steps and the answer (with tool findings)', async () => { + const { model, calls } = scriptedModel((info) => { + const stage = stageOf(info) + if (stage === 'planner') return { text: plan(['Find the id', 'Use it']) } + if (stage === 'executor') { + if (info.user.includes('STEP 1/2')) { + return executorReply(info, [{ name: 'list_notes', args: {} }], 'Found note id 4711.') + } + return { text: 'Used it.' } + } + return { text: 'The id is 4711.' } + }) + const agent = await createAgent({ model, tools: noteTools([]) }) + await agent.run('find and use the id') + const step2 = calls + .filter((c) => stageOf(c) === 'executor') + .find((c) => c.user.includes('STEP 2/2'))! + assert.match(step2.user, /Found note id 4711/) + const synth = calls.find((c) => stageOf(c) === 'synthesizer')! + assert.match(synth.user, /FINDINGS \(tool results\):\n- list_notes: \{"notes":\[\]\}/) +}) + +// ── subagents ─────────────────────────────────────────────────────────────── + +test('subagent (in-process): the child runs the task and its usage is charged to the parent', async () => { + const child = scriptedModel((info) => + stageOf(info) === 'planner' + ? { text: plan(['Research']) } + : stageOf(info) === 'executor' + ? { text: 'researched' } + : { text: 'CHILD-ANSWER' }, + ) + const parent = scriptedModel((info) => { + const stage = stageOf(info) + if (stage === 'planner') return { text: plan(['Delegate']) } + if (stage === 'executor') { + return executorReply(info, [{ name: 'research', args: { task: 'look it up' } }], 'Delegated.') + } + return { text: 'Parent answer.' } + }) + const agent = await createAgent({ + model: parent.model, + tools: { + research: createSubagentTool({ + name: 'researcher', + description: 'Research a question', + config: { model: child.model }, + }), + }, + }) + const { events, onEvent } = collect() + const result = await agent.run('research something', { onEvent }) + const call = result.trace[0].toolCalls[0] + assert.equal(call.ok, true) + assert.equal(call.output, 'CHILD-ANSWER') + assert.ok(events.some((e) => e.type === 'subagent.start' && e.task === 'look it up')) + assert.ok(events.some((e) => e.type === 'subagent.event' && e.event.type === 'plan.created')) + assert.ok(events.some((e) => e.type === 'usage' && e.phase === 'subagent')) + const complete = events.find((e) => e.type === 'subagent.complete') + assert.ok(complete?.type === 'subagent.complete' && complete.usage.totalTokens > 0) +}) + +test('subagent (worker): runs behind a message channel, calls a proxied host tool through the parent gate', async () => { + const childModel = scriptedModel((info) => { + const stage = stageOf(info) + if (stage === 'planner') return { text: plan(['Use host tool']) } + if (stage === 'executor') + return executorReply(info, [{ name: 'add_note', args: { text: 'from-worker' } }]) + return { text: 'WORKER-DONE' } + }) + const opened: MessagePort[] = [] + const makeWorker = () => { + const channel = new MessageChannel() + opened.push(channel.port1, channel.port2) + serveSubagentWorker({ scope: channel.port2, resolveModel: () => childModel.model }) + channel.port1.start() + channel.port2.start() + return Object.assign(channel.port1, { + terminate: () => { + channel.port1.close() + channel.port2.close() + }, + }) + } + const log: string[] = [] + const parent = scriptedModel((info) => { + const stage = stageOf(info) + if (stage === 'planner') return { text: plan(['Delegate']) } + if (stage === 'executor') + return executorReply(info, [{ name: 'worker', args: { task: 'add a note' } }]) + return { text: 'ok' } + }) + const asked: string[] = [] + const agent = await createAgent({ + model: parent.model, + tools: { + worker: createSubagentTool({ + name: 'w', + description: 'Delegate to a worker', + worker: makeWorker, + workerConfig: { model: { providerType: 'openai', model: 'x', apiKey: 'k' } }, + tools: { add_note: noteTools(log).add_note }, + }), + }, + toolApproval: { + mode: 'ask-writes', + onRequest: (req) => { + asked.push(req.toolName) + return true + }, + }, + }) + const { events, onEvent } = collect() + const result = await agent.run('delegate a note', { onEvent }) + assert.equal(result.trace[0].toolCalls[0].output, 'WORKER-DONE') + assert.deepEqual(log, ['add:from-worker']) + // The parent gate saw both the delegation and the host tool the child called. + assert.deepEqual(asked, ['worker', 'add_note']) + assert.ok(events.some((e) => e.type === 'subagent.event' && e.event.type === 'step.tool-call')) + for (const p of opened) p.close() +}) diff --git a/tests/files-images.test.ts b/tests/files-images.test.ts new file mode 100644 index 0000000..d7dc159 --- /dev/null +++ b/tests/files-images.test.ts @@ -0,0 +1,266 @@ +import 'fake-indexeddb/auto' +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { + AttachmentsNotSupportedError, + attachmentKind, + createAgent, + supportsPdf, + toFilePart, + createFileTools, + ImagesNotSupportedError, + normalizePath, + supportsImages, + VirtualFileSystem, + type AgentEvent, +} from '../dist/index.js' +import { scriptedModel, stageOf, type CallInfo } from './helpers/scripted-model.ts' + +const PNG = + 'data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8/5+hHgAHggJ/PchI7wAAAABJRU5ErkJggg==' + +const callTool = (tools: Record, name: string, args: unknown) => + (tools[name] as { execute: (a: unknown, o: unknown) => Promise }).execute(args, {}) + +// ── virtual file system ───────────────────────────────────────────────────── + +test('normalizePath: absolute, collapsed, and never above the root', () => { + assert.equal(normalizePath('notes/./a//b.md'), '/notes/a/b.md') + assert.equal(normalizePath('/x/../y.txt'), '/y.txt') + assert.throws(() => normalizePath('../etc/passwd'), /escapes the root/) + assert.throws(() => normalizePath('/'), /required/) +}) + +for (const mode of ['memory', 'indexeddb'] as const) { + test(`VirtualFileSystem (${mode}): write, read, list by prefix, data URLs, delete, change events`, async () => { + const vfs = new VirtualFileSystem( + mode === 'memory' ? { memory: true } : { dbName: 'vfs-test', namespace: 'n1' }, + ) + const changes: string[] = [] + vfs.onChange((c) => changes.push(`${c.type}:${c.path}`)) + await vfs.write('/docs/a.md', '# A') + await vfs.write('docs/b.txt', 'bee') + await vfs.writeDataUrl('/img/dot.png', PNG) + assert.deepEqual( + (await vfs.list()).map((f) => f.path), + ['/docs/a.md', '/docs/b.txt', '/img/dot.png'], + ) + assert.deepEqual( + (await vfs.list('/docs')).map((f) => f.path), + ['/docs/a.md', '/docs/b.txt'], + ) + const a = await vfs.read('/docs/a.md') + assert.equal(a?.content, '# A') + assert.equal(a?.mimeType, 'text/markdown') + const img = await vfs.read('/img/dot.png') + assert.equal(img?.encoding, 'base64') + assert.equal(img?.mimeType, 'image/png') + assert.equal(await vfs.readDataUrl('/img/dot.png'), PNG) + assert.equal(await vfs.delete('/docs/b.txt'), true) + assert.equal(await vfs.delete('/docs/b.txt'), false) + assert.equal(await vfs.read('/docs/b.txt'), undefined) + assert.deepEqual(changes, [ + 'write:/docs/a.md', + 'write:/docs/b.txt', + 'write:/img/dot.png', + 'delete:/docs/b.txt', + ]) + await vfs.clear() + assert.deepEqual(await vfs.list(), []) + }) +} + +test('VirtualFileSystem: namespaces in one database do not see each other', async () => { + const a = new VirtualFileSystem({ dbName: 'vfs-ns', namespace: 'a' }) + const b = new VirtualFileSystem({ dbName: 'vfs-ns', namespace: 'b' }) + await a.write('/x.txt', 'from a') + assert.deepEqual(await b.list(), []) + assert.equal((await a.list()).length, 1) +}) + +test('VirtualFileSystem: refuses a file above the size cap', async () => { + const vfs = new VirtualFileSystem({ memory: true, maxFileBytes: 4 }) + await assert.rejects(() => vfs.write('/big.txt', 'hello'), /limit is 4/) +}) + +test('createFileTools: list / read / write / delete, read-only tools marked, binary refused as text', async () => { + const vfs = new VirtualFileSystem({ memory: true }) + const tools = createFileTools(vfs) + assert.deepEqual(Object.keys(tools), ['fs_list', 'fs_read', 'fs_write', 'fs_delete']) + assert.equal((tools.fs_read as { readOnly?: boolean }).readOnly, true) + assert.equal((tools.fs_write as { readOnly?: boolean }).readOnly, undefined) + assert.deepEqual(await callTool(tools, 'fs_write', { path: '/r.md', content: 'report' }), { + written: '/r.md', + size: 6, + }) + assert.equal(await callTool(tools, 'fs_read', { path: '/r.md' }), 'report') + await assert.rejects(() => callTool(tools, 'fs_read', { path: '/nope' }), /files: \/r.md/) + await vfs.writeDataUrl('/p.png', PNG) + assert.match(String(await callTool(tools, 'fs_read', { path: '/p.png' })), /binary file/) + assert.deepEqual(await callTool(tools, 'fs_delete', { path: '/r.md' }), { deleted: true }) + assert.deepEqual(Object.keys(createFileTools(vfs, { readOnly: true, prefix: 'ws_' })), [ + 'ws_list', + 'ws_read', + ]) +}) + +// ── images ────────────────────────────────────────────────────────────────── + +test('supportsImages: local runtimes and text-only families say no, vision providers yes', () => { + assert.equal(supportsImages({ provider: 'web-llm', modelId: 'Qwen3' }), false) + assert.equal(supportsImages({ provider: 'deepseek.chat', modelId: 'deepseek-chat' }), false) + assert.equal(supportsImages({ provider: 'openai.chat', modelId: 'gpt-3.5-turbo' }), false) + assert.equal( + supportsImages({ provider: 'google.generative-ai', modelId: 'gemini-3.5-flash' }), + true, + ) + assert.equal(supportsImages({ provider: 'anthropic.messages', modelId: 'claude-sonnet-5' }), true) + assert.equal(supportsImages('openai/gpt-5.4-mini'), true) + assert.equal(supportsImages({ provider: 'my-server.chat', modelId: 'llama' }), undefined) +}) + +const imageParts = (info: CallInfo): unknown[] => + ((info.options.prompt ?? []) as { role: string; content: unknown }[]) + .filter((m) => m.role === 'user' && Array.isArray(m.content)) + .flatMap((m) => + (m.content as { type: string }[]).filter((p) => p.type === 'file' || p.type === 'image'), + ) + +test('images reach the planner, the executor and the synthesizer of a vision model', async () => { + const { model, calls } = scriptedModel((info) => + stageOf(info) === 'planner' + ? { text: JSON.stringify({ thought: 't', steps: [{ description: 'Describe it' }] }) } + : { text: 'A red dot.' }, + ) + const agent = await createAgent({ model, vision: true }) + assert.equal(agent.capabilities.images, true) + const result = await agent.run('what is in the picture?', { + images: [{ data: PNG, name: 'dot.png' }], + }) + assert.equal(result.final, 'A red dot.') + for (const stage of ['planner', 'executor', 'synthesizer']) { + const call = calls.find((c) => stageOf(c) === stage)! + assert.equal(imageParts(call).length, 1, `${stage} sees the image`) + } +}) + +test('images + a text-only model: the run ends at once with a clear error, no tokens spent', async () => { + const { model, calls } = scriptedModel(() => ({ text: 'unused' })) + const agent = await createAgent({ model, toolMode: 'prompted' }) + assert.equal(agent.capabilities.images, false) + const events: AgentEvent[] = [] + const result = await agent.run('describe this', { + images: [{ data: PNG, name: 'dot.png' }], + onEvent: (e) => events.push(e), + }) + assert.equal(calls.length, 0) + assert.match(result.final, /can't take images \(it runs in the text-only prompted tool mode\)/) + const err = events.find((e) => e.type === 'error') + assert.ok(err?.type === 'error' && err.phase === 'run') + // `vision: false` forces the same for a model that would otherwise be tried. + const forced = await createAgent({ model, vision: false }) + assert.match((await forced.run('look', { images: [{ data: PNG }] })).final, /can't take images/) +}) + +test('images + a provider that refuses them: the refusal becomes ImagesNotSupportedError', async () => { + const { model } = scriptedModel((info) => { + if (imageParts(info).length) throw new Error('This model does not support image input') + return { text: 'ok' } + }) + const agent = await createAgent({ model }) + assert.equal(agent.capabilities.images, undefined) + const events: AgentEvent[] = [] + const result = await agent.run('what is this', { + images: [{ data: PNG }], + onEvent: (e) => events.push(e), + }) + assert.match( + result.final, + /can't take images \(the provider said: This model does not support image input\)/, + ) + assert.ok(events.some((e) => e.type === 'error' && e.phase === 'run')) + assert.ok(new ImagesNotSupportedError('m') instanceof Error) +}) + +test('memory keeps a text note of the images, not their bytes', async () => { + const { MemoryStore } = await import('../dist/index.js') + const memory = new MemoryStore() + const { model } = scriptedModel((info) => + stageOf(info) === 'planner' + ? { text: JSON.stringify({ thought: 'Hello!', steps: [] }) } + : { text: 'x' }, + ) + const agent = await createAgent({ model, memory, sessionId: 's', vision: true }) + await agent.run('hi', { images: [{ data: PNG, name: 'cat.png' }] }) + const stored = await memory.load('s') + assert.match(stored[0].content, /\[attached: cat.png\]/) + assert.doesNotMatch(stored[0].content, /base64/) +}) + +// ── PDFs, other files, URLs ───────────────────────────────────────────────── + +const PDF = 'data:application/pdf;base64,JVBERi0xLjQKJcfsj6IK' + +test('attachments: kind and part by media type — image part, PDF file part, URL passthrough', () => { + assert.equal(attachmentKind({ data: PNG }), 'image') + assert.equal(attachmentKind({ data: PDF }), 'pdf') + assert.equal(attachmentKind({ data: 'https://example.com/report.pdf' }), 'pdf') + assert.equal(attachmentKind({ data: 'aGk=', mediaType: 'text/csv' }), 'file') + assert.deepEqual(toFilePart({ data: PDF, name: 'r.pdf' }), { + type: 'file', + data: 'JVBERi0xLjQKJcfsj6IK', + mediaType: 'application/pdf', + filename: 'r.pdf', + }) + const url = toFilePart({ data: 'https://example.com/cat.png' }) as { + type: string + data: URL + mediaType: string + } + assert.equal(url.type, 'file') + assert.equal(url.mediaType, 'image/png') + assert.ok(url.data instanceof URL) + assert.equal(url.data.href, 'https://example.com/cat.png') +}) + +test('supportsPdf: Gemini/Claude/OpenAI read PDFs; local and text-only models do not', () => { + assert.equal(supportsPdf({ provider: 'google.generative-ai', modelId: 'gemini-3.5-flash' }), true) + assert.equal(supportsPdf({ provider: 'anthropic.messages', modelId: 'claude-haiku-4-5' }), true) + assert.equal(supportsPdf({ provider: 'web-llm', modelId: 'x' }), false) + assert.equal(supportsPdf({ provider: 'xai.chat', modelId: 'grok-4' }), undefined) +}) + +test('a PDF reaches the model as a file part; a URL stays a URL', async () => { + const { model, calls } = scriptedModel((info) => + stageOf(info) === 'planner' + ? { text: JSON.stringify({ thought: 't', steps: [{ description: 'Read it' }] }) } + : { text: 'It is a report.' }, + ) + const agent = await createAgent({ model, inputs: { pdf: true, images: true } }) + await agent.run('summarise the attachments', { + files: [{ data: PDF, name: 'r.pdf' }, { data: 'https://example.com/cat.png' }], + }) + const parts = imageParts(calls.find((c) => stageOf(c) === 'planner')!) as { + type: string + mediaType: string + data: unknown + }[] + assert.deepEqual( + parts.map((p) => p.mediaType), + ['application/pdf', 'image/png'], + ) + // The provider receives the link itself, not downloaded bytes. + const link = parts[1].data as { type: string; url: URL } + assert.equal(link.type, 'url') + assert.equal(String(link.url), 'https://example.com/cat.png') +}) + +test('a PDF for a model that cannot read PDFs: a clear error that suggests converting it', async () => { + const { model, calls } = scriptedModel(() => ({ text: 'unused' })) + const agent = await createAgent({ model, inputs: { pdf: false } }) + assert.equal(agent.capabilities.pdf, false) + const result = await agent.run('read this', { files: [{ data: PDF, name: 'r.pdf' }] }) + assert.equal(calls.length, 0) + assert.match(result.final, /can't take PDF files\. Convert the PDF to text first/) + assert.equal(new AttachmentsNotSupportedError('m', 'pdf').kind, 'pdf') +}) diff --git a/tests/helpers/scripted-model.ts b/tests/helpers/scripted-model.ts new file mode 100644 index 0000000..8127a9b --- /dev/null +++ b/tests/helpers/scripted-model.ts @@ -0,0 +1,153 @@ +import { simulateReadableStream } from 'ai' +import { MockLanguageModelV4 } from 'ai/test' + +/** What the scripted model answers for one call. */ +export interface Reply { + text?: string + reasoning?: string + toolCalls?: { name: string; args: unknown }[] + usage?: { input?: number; output?: number; reasoning?: number; cacheRead?: number } +} + +/** A readable view of one model call, for routing and assertions. */ +export interface CallInfo { + /** Concatenated system message text. */ + system: string + /** Concatenated user message text. */ + user: string + /** Number of tool results already in the conversation (multi-step position). */ + toolResults: number + /** Names of the tools offered to the model on this call. */ + tools: string[] + /** The raw call options the SDK passed to the model. */ + options: Record +} + +type Part = { type: string; text?: string } +type Message = { role: string; content: string | Part[]; providerOptions?: unknown } + +const textOf = (content: Message['content']): string => + typeof content === 'string' + ? content + : content + .map((p) => (p.type === 'text' ? (p.text ?? '') : '')) + .filter(Boolean) + .join('\n') + +export const infoOf = (options: Record): CallInfo => { + const prompt = (options.prompt ?? []) as Message[] + return { + system: prompt + .filter((m) => m.role === 'system') + .map((m) => textOf(m.content)) + .join('\n'), + user: prompt + .filter((m) => m.role === 'user') + .map((m) => textOf(m.content)) + .join('\n'), + toolResults: prompt.filter((m) => m.role === 'tool').length, + tools: ((options.tools ?? []) as { name: string }[]).map((t) => t.name), + options, + } +} + +const usageOf = (r: Reply) => { + const input = r.usage?.input ?? 10 + const output = r.usage?.output ?? 5 + return { + inputTokens: { + total: input, + noCache: input - (r.usage?.cacheRead ?? 0), + cacheRead: r.usage?.cacheRead ?? 0, + cacheWrite: 0, + }, + outputTokens: { + total: output, + text: output - (r.usage?.reasoning ?? 0), + reasoning: r.usage?.reasoning ?? 0, + }, + } +} + +let callSeq = 0 + +/** + * A MockLanguageModelV4 whose every call is answered by `route(info)`. Both + * doGenerate (structured output, prompted rounds) and doStream (native tool + * loop, synthesizer) are implemented, including reasoning and tool calls. + */ +export const scriptedModel = (route: (info: CallInfo) => Reply) => { + const calls: CallInfo[] = [] + const model = new MockLanguageModelV4({ + provider: 'scripted', + modelId: 'scripted-1', + // Accept links as-is (like Gemini / Claude), so the SDK never downloads them. + supportedUrls: { '*/*': [/^https?:\/\//] }, + doGenerate: async (options: Record) => { + const info = infoOf(options) + calls.push(info) + const r = route(info) + const content: unknown[] = [] + if (r.reasoning) content.push({ type: 'reasoning', text: r.reasoning }) + if (r.text !== undefined) content.push({ type: 'text', text: r.text }) + for (const c of r.toolCalls ?? []) { + content.push({ + type: 'tool-call', + toolCallId: `call-${(callSeq += 1)}`, + toolName: c.name, + input: JSON.stringify(c.args ?? {}), + }) + } + return { + content, + finishReason: { unified: r.toolCalls?.length ? 'tool-calls' : 'stop', raw: undefined }, + usage: usageOf(r), + warnings: [], + } + }, + doStream: async (options: Record) => { + const info = infoOf(options) + calls.push(info) + const r = route(info) + const chunks: unknown[] = [{ type: 'stream-start', warnings: [] }] + if (r.reasoning) { + chunks.push( + { type: 'reasoning-start', id: 'r1' }, + { type: 'reasoning-delta', id: 'r1', delta: r.reasoning }, + { type: 'reasoning-end', id: 'r1' }, + ) + } + if (r.text) { + chunks.push( + { type: 'text-start', id: 't1' }, + { type: 'text-delta', id: 't1', delta: r.text }, + { type: 'text-end', id: 't1' }, + ) + } + for (const c of r.toolCalls ?? []) { + chunks.push({ + type: 'tool-call', + toolCallId: `call-${(callSeq += 1)}`, + toolName: c.name, + input: JSON.stringify(c.args ?? {}), + }) + } + chunks.push({ + type: 'finish', + finishReason: { unified: r.toolCalls?.length ? 'tool-calls' : 'stop', raw: undefined }, + usage: usageOf(r), + }) + return { stream: simulateReadableStream({ chunks }) } + }, + } as never) + return { model, calls } +} + +/** Which agent stage a call belongs to (by the role marker in its system prompt). */ +export const stageOf = (info: CallInfo): string => { + if (info.system.includes('REPLANNER')) return 'replanner' + if (info.system.includes('PLANNER')) return 'planner' + if (info.system.includes('EXECUTOR')) return 'executor' + if (info.system.includes('SYNTHESIZER')) return 'synthesizer' + return 'other' +} diff --git a/tests/mcp.test.ts b/tests/mcp.test.ts index bd5e3f0..82bd00a 100644 --- a/tests/mcp.test.ts +++ b/tests/mcp.test.ts @@ -34,8 +34,12 @@ interface RpcMessage { interface MockOptions { requireAuth?: boolean - tools?: { name: string; description?: string; inputSchema?: unknown }[] + tools?: { name: string; description?: string; inputSchema?: unknown; annotations?: unknown }[] failListTools?: boolean + /** Paginate tools/list with this many tools per page. */ + pageSize?: number + /** Never answer tools/list (a hung server). */ + hangListTools?: boolean } const DEFAULT_TOOLS = [ @@ -97,6 +101,18 @@ const createMockServer = (opts: MockOptions = {}) => { error: { code: -32000, message: 'tools/list exploded' }, } } + if (opts.pageSize) { + const start = Number((msg.params as { cursor?: string } | undefined)?.cursor ?? 0) + const end = start + opts.pageSize + return { + jsonrpc: '2.0', + id: msg.id, + result: { + tools: tools.slice(start, end), + ...(end < tools.length ? { nextCursor: String(end) } : {}), + }, + } + } return { jsonrpc: '2.0', id: msg.id, result: { tools } } } if (msg.method === 'tools/call') { @@ -231,6 +247,9 @@ const createMockServer = (opts: MockOptions = {}) => { } const parsed = JSON.parse(String(init.body ?? 'null')) as RpcMessage | RpcMessage[] const messages = Array.isArray(parsed) ? parsed : [parsed] + if (opts.hangListTools && messages.some((m) => m.method === 'tools/list')) { + return new Promise(() => {}) + } const replies = messages.map(handleRpc).filter((r) => r !== undefined) if (replies.length === 0) return new Response(null, { status: 202 }) return jsonResponse(replies.length === 1 ? replies[0] : replies) @@ -947,3 +966,63 @@ test('connectMcpHttp: an ordinary server error is not dressed up as a CORS probl assert.doesNotMatch(docs?.error ?? '', /Access-Control/) await mcp.close() }) + +// ── large catalogues: pagination, deadlines, annotations ──────────────────── + +test('connectMcpHttp: follows nextCursor so every page of tools is mounted', async () => { + const tools = Array.from({ length: 25 }, (_, i) => ({ + name: `t${i}`, + inputSchema: { type: 'object' }, + })) + const mock = createMockServer({ tools, pageSize: 10 }) + const mcp = await connectMcpHttp({ big: { url: MCP_URL, fetch: mock.fetchFn } }) + assert.equal(Object.keys(mcp.tools).length, 25) + assert.equal(mcp.catalog.at(-1)?.name, 'big__t24') + await mcp.refreshServer('big') + assert.equal(Object.keys(mcp.tools).length, 25) + await mcp.close() +}) + +test('connectMcpHttp: a server that hangs is cut off by connectTimeoutMs, others still mount', async () => { + const hung = createMockServer({ hangListTools: true }) + const fine = createMockServer() + const started = Date.now() + const mcp = await connectMcpHttp( + { + slow: { url: MCP_URL, fetch: hung.fetchFn }, + docs: { url: MCP_URL, fetch: fine.fetchFn }, + }, + { connectTimeoutMs: 300 }, + ) + assert.ok(Date.now() - started < 3_000) + assert.deepEqual( + mcp.results.map((r) => [r.name, r.connected]), + [ + ['slow', false], + ['docs', true], + ], + ) + assert.match(mcp.results[0].error ?? '', /timed out after 300 ms/) + assert.deepEqual(Object.keys(mcp.tools), ['docs__echo']) + await mcp.close() +}) + +test('connectMcpHttp: readOnlyHint marks tools read-only for the consent gate', async () => { + const mock = createMockServer({ + tools: [ + { name: 'read', inputSchema: { type: 'object' }, annotations: { readOnlyHint: true } }, + { name: 'write', inputSchema: { type: 'object' } }, + ], + }) + const mcp = await connectMcpHttp({ s: { url: MCP_URL, fetch: mock.fetchFn } }) + assert.equal((mcp.tools.s__read as { readOnly?: boolean }).readOnly, true) + assert.equal((mcp.tools.s__write as { readOnly?: boolean }).readOnly, undefined) + assert.deepEqual( + mcp.catalog.map((c) => [c.name, c.readOnly ?? false]), + [ + ['s__read', true], + ['s__write', false], + ], + ) + await mcp.close() +}) diff --git a/tests/token-efficiency.test.ts b/tests/token-efficiency.test.ts new file mode 100644 index 0000000..be36190 --- /dev/null +++ b/tests/token-efficiency.test.ts @@ -0,0 +1,182 @@ +import { test } from 'node:test' +import assert from 'node:assert/strict' +import { z } from 'zod' +import { + createAgent, + createToolResultClearer, + defineTool, + withRollingBreakpoint, + type AgentEvent, +} from '../dist/index.js' +import { scriptedModel, stageOf, type CallInfo, type Reply } from './helpers/scripted-model.ts' + +type RawMessage = { role: string; content: unknown; providerOptions?: unknown } +type RawPart = { type: string; toolName?: string; output?: { type: string; value: unknown } } + +const plan = (steps: string[]) => + JSON.stringify({ thought: 'plan', steps: steps.map((description) => ({ description })) }) + +const BREAKPOINT = { anthropic: { cacheControl: { type: 'ephemeral' } } } + +test('withRollingBreakpoint marks only the last message and keeps its own options', () => { + const msgs = [ + { role: 'user', content: 'a' }, + { role: 'assistant', content: 'b', providerOptions: { openai: { x: 1 } } }, + ] + const out = withRollingBreakpoint(msgs, { enabled: true, ttl: '1h' }) + assert.equal(out[0].providerOptions, undefined) + assert.deepEqual(out[1].providerOptions, { + openai: { x: 1 }, + anthropic: { cacheControl: { type: 'ephemeral', ttl: '1h' } }, + }) + assert.equal(msgs[1].providerOptions.openai.x, 1) // input untouched + assert.equal(withRollingBreakpoint(msgs, { enabled: false }), msgs) + // A breakpoint on an earlier message moves to the newest one. + const again = withRollingBreakpoint([...out, { role: 'user', content: 'c' }], { enabled: true }) + assert.deepEqual(again[1].providerOptions, { openai: { x: 1 } }) + assert.deepEqual(again[2].providerOptions, BREAKPOINT) +}) + +test('createToolResultClearer: clears the oldest results past the trigger, sticky, keeps the last N', () => { + const infos: unknown[] = [] + const clear = createToolResultClearer({ triggerTokens: 30, keep: 1 }, (i) => infos.push(i)) + const result = (id: string, value: string) => + ({ + role: 'tool', + content: [ + { type: 'tool-result', toolCallId: id, toolName: 'read', output: { type: 'text', value } }, + ], + }) as never + const big = 'x'.repeat(200) + const round1 = [{ role: 'user', content: 'go' } as never, result('a', 'small')] + assert.equal(clear(round1), round1) // under the trigger: untouched + const round2 = [...round1, result('b', big), result('c', big)] + const edited = clear(round2) as unknown as { content: RawPart[] }[] + const value = (i: number) => edited[i].content[0].output?.value as string + assert.match(value(1), /read result cleared/) + assert.match(value(2), /read result cleared/) + assert.equal(value(3), big) // the newest result stays + assert.equal(infos.length, 1) + // Sticky: the same ids stay cleared even once the context is small again. + const shrunk = clear([round1[0], result('a', 'small')]) as unknown as { content: RawPart[] }[] + assert.match(shrunk[1].content[0].output?.value as string, /cleared/) +}) + +test('tool loop: the cache breakpoint rolls to the newest message every round; tools keep declaration order', async () => { + const route = (info: CallInfo): Reply => { + if (stageOf(info) === 'planner') return { text: plan(['Look it up']) } + if (stageOf(info) === 'executor') { + return info.toolResults === 0 + ? { toolCalls: [{ name: 'lookup', args: { q: 'x' } }] } + : { text: 'found it' } + } + return { text: 'answer' } + } + const { model, calls } = scriptedModel(route) + const agent = await createAgent({ + model, + tools: { + zeta: defineTool({ description: 'z', inputSchema: z.object({}), execute: async () => 'z' }), + lookup: defineTool({ + description: 'look up', + inputSchema: z.object({ q: z.string() }), + execute: async () => 'value', + }), + }, + }) + await agent.run('find x') + const exec = calls.filter((c) => stageOf(c) === 'executor') + assert.equal(exec.length, 2) + assert.deepEqual(exec[0].tools, ['zeta', 'lookup']) + for (const call of exec) { + const prompt = call.options.prompt as RawMessage[] + const marked = prompt.filter((m) => + JSON.stringify(m.providerOptions ?? {}).includes('cacheControl'), + ) + // The system message and the newest message — nothing in between. + assert.equal(prompt[0].role, 'system') + assert.deepEqual(prompt.at(-1)?.providerOptions, BREAKPOINT) + assert.equal(marked.length, 2) + } + assert.equal((exec[1].options.prompt as RawMessage[]).at(-1)?.role, 'tool') +}) + +test('tool loop: stale tool results are cleared past the threshold and the clearing is reported', async () => { + const big = 'y'.repeat(4000) + let round = 0 + const route = (info: CallInfo): Reply => { + if (stageOf(info) === 'planner') return { text: plan(['Read three pages']) } + if (stageOf(info) === 'executor') { + round = info.toolResults + return round < 3 + ? { toolCalls: [{ name: 'page', args: { n: round } }] } + : { text: 'read all' } + } + return { text: 'answer' } + } + const { model, calls } = scriptedModel(route) + const events: AgentEvent[] = [] + const agent = await createAgent({ + model, + maxStepsPerTask: 6, + compaction: { clearToolResultsAfterTokens: 1200, keepToolResults: 1 }, + tools: { + page: defineTool({ + description: 'read a page', + inputSchema: z.object({ n: z.number() }), + execute: async () => big, + }), + }, + }) + await agent.run('read', { onEvent: (e) => events.push(e) }) + assert.equal(events.filter((e) => e.type === 'step.tool-result').length, 3) + const last = calls.filter((c) => stageOf(c) === 'executor').at(-1)! + const outputs = (last.options.prompt as RawMessage[]) + .filter((m) => m.role === 'tool') + .flatMap((m) => m.content as RawPart[]) + .map((p) => String(p.output?.value)) + assert.equal(outputs.length, 3) + assert.match(outputs[0], /page result cleared to save context/) + assert.match(outputs[1], /page result cleared to save context/) + assert.equal(outputs[2], big) + const compacted = events.filter( + (e) => e.type === 'context.compacted' && e.scope === 'tool-results', + ) + assert.ok(compacted.length >= 1) +}) + +test('promptCaching: false → no rolling breakpoint in the tool loop', async () => { + const route = (info: CallInfo): Reply => { + if (stageOf(info) === 'planner') return { text: plan(['Do it']) } + return { text: 'ok' } + } + const { model, calls } = scriptedModel(route) + await (await createAgent({ model, promptCaching: false })).run('do it now please') + for (const call of calls) { + assert.ok(!JSON.stringify(call.options.prompt).includes('cacheControl')) + } +}) + +test('createToolResultClearer: works by position, so repeated tool call ids are fine', () => { + const clear = createToolResultClearer({ triggerTokens: 30, keep: 1 }) + const result = (value: string) => + ({ + role: 'tool', + content: [ + { + type: 'tool-result', + toolCallId: 'call_0', + toolName: 'read', + output: { type: 'text', value }, + }, + ], + }) as never + const big = 'x'.repeat(200) + const out = clear([result(big), result(big), result(big)]) as unknown as { + content: RawPart[] + }[] + const values = out.map((m) => String(m.content[0].output?.value)) + assert.match(values[0], /cleared/) + assert.match(values[1], /cleared/) + assert.equal(values[2], big) +}) diff --git a/tsconfig.json b/tsconfig.json index fa80e93..0c55ac8 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -14,7 +14,8 @@ "skipLibCheck": true, "forceConsistentCasingInFileNames": true, "verbatimModuleSyntax": true, - "outDir": "dist" + "outDir": "dist", + "ignoreDeprecations": "6.0" }, "include": ["src"] }