From 8e23569d0fdbb8ac26c18f8f7acb5217689dc812 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 2 Oct 2026 23:51:59 +0000 Subject: [PATCH 1/2] feat: thinking, token limits, compaction, skills, tool consent, tool search, subagents, token efficiency MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Shared spec with @dudko.dev/agent-web (same config fields, events and semantics): - Thinking: `thinking` / `stageThinking` → the AI SDK's portable `reasoning` plus provider budgets; thoughts stream as step/final reasoning-delta events. - Limits: run-level token caps (input / output / reasoning / total), per-call output caps per stage, `maxToolCalls`, `maxPlanSteps`; a crossed cap stops at the next boundary, emits budget.exceeded and still answers. - Compaction: auto (history at run start, trace before each call) and manual `agent.compact()`; per-result tool-output caps; stale tool results inside a step's loop become stubs past `clearToolResultsAfterTokens` (sticky, position-keyed). - Skills (agentskills.io SKILL.md): index in the prompt, body on activation by the planner or `load_skill`; `read_skill_file`. - Tool consent: autopilot / ask-writes / ask-all / read-only, glob rules, remember, timeout; switchable live (setToolApprovalMode). - Large catalogues: paginated tools/list, per-server connect timeout, 'auto' tool strategy → `find_tools` search above the threshold. - Prompt caching: run-stable system prompts with an Anthropic breakpoint, OpenAI promptCacheKey, a rolling breakpoint on the newest loop message, tools sorted by name. - Subagents: createSubagentTool in worker_threads or in-process, host tools proxied through the parent's consent gate. - Autonomy rules in the prompts: act, don't ask. - CLI flags/config for all of it; README, CLAUDE.md, env.example; deps bumped. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01A7jHa5Pim4ufrdgxg561G5 --- CLAUDE.md | 19 +- README.md | 283 ++++++++++- env.example | 54 +- package-lock.json | 174 +++---- package.json | 22 +- src/agent.ts | 619 +++++++++++++++++++++++ src/approval.ts | 331 +++++++++++++ src/caching.ts | 108 ++++ src/call-options.ts | 50 ++ src/cli/args.ts | 20 +- src/cli/config.ts | 108 +++- src/cli/run.ts | 287 +++++++++-- src/compaction.ts | 277 +++++++++++ src/context-editing.ts | 95 ++++ src/context.ts | 11 + src/executor.ts | 181 ++++++- src/index.ts | 505 +++---------------- src/internal.ts | 38 +- src/limits.ts | 78 +++ src/mcp.ts | 157 +++++- src/planner.ts | 98 +++- src/prompts.ts | 235 +++++++-- src/replanner.ts | 40 +- src/runner.ts | 208 +++++++- src/skills.ts | 394 +++++++++++++++ src/subagent-worker.ts | 100 ++++ src/subagent.ts | 440 +++++++++++++++++ src/synthesizer.ts | 36 +- src/thinking.ts | 139 ++++++ src/tool-search.ts | 184 +++++++ src/tool-wrap.ts | 110 +++++ src/types.ts | 267 +++++++++- src/utils.ts | 68 +++ tests/approval.test.ts | 192 ++++++++ tests/compaction.test.ts | 252 ++++++++++ tests/config.test.ts | 68 ++- tests/dist-loadable.test.ts | 32 ++ tests/features-integration.test.ts | 760 +++++++++++++++++++++++++++++ tests/helpers/mcp-server.ts | 1 + tests/helpers/openai-server.ts | 57 ++- tests/limits.test.ts | 105 ++++ tests/mcp-connect.test.ts | 112 ++++- tests/skills.test.ts | 184 +++++++ tests/subagent.test.ts | 72 +++ tests/thinking.test.ts | 144 ++++++ tests/token-efficiency.test.ts | 132 +++++ tests/tool-search.test.ts | 99 ++++ tsup.config.ts | 13 + 48 files changed, 7194 insertions(+), 765 deletions(-) create mode 100644 src/agent.ts create mode 100644 src/approval.ts create mode 100644 src/caching.ts create mode 100644 src/call-options.ts create mode 100644 src/compaction.ts create mode 100644 src/context-editing.ts create mode 100644 src/limits.ts create mode 100644 src/skills.ts create mode 100644 src/subagent-worker.ts create mode 100644 src/subagent.ts create mode 100644 src/thinking.ts create mode 100644 src/tool-search.ts create mode 100644 src/tool-wrap.ts create mode 100644 tests/approval.test.ts create mode 100644 tests/compaction.test.ts create mode 100644 tests/features-integration.test.ts create mode 100644 tests/limits.test.ts create mode 100644 tests/skills.test.ts create mode 100644 tests/subagent.test.ts create mode 100644 tests/thinking.test.ts create mode 100644 tests/token-efficiency.test.ts create mode 100644 tests/tool-search.test.ts diff --git a/CLAUDE.md b/CLAUDE.md index 88678de..20b44aa 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -7,6 +7,17 @@ ## Layout - `src/` — library + CLI source (`.ts`, imported with explicit `.ts` extensions) + - `agent.ts` (`createAgent`, `IAgent`), `runner.ts` (the loop), `planner.ts` / + `executor.ts` / `replanner.ts` / `synthesizer.ts`, `prompts.ts` (system = + run-stable, user = dynamic), `call-options.ts` (per-stage SDK options). + - Feature modules: `thinking.ts`, `limits.ts`, `compaction.ts`, `skills.ts`, + `approval.ts`, `tool-wrap.ts` (approval gate + tool-call budget + output + cap, wrapped around every tool's `execute`), `tool-search.ts`, + `caching.ts` (system breakpoint + rolling message breakpoint), + `context-editing.ts` (stale tool results → stubs inside a step's loop), + `subagent.ts` + `subagent-worker.ts` (worker entry, built to + `dist/subagent-worker.js`). + - `index.ts` is exports only. - `tests/` — `node:test` suites, run via `node --experimental-strip-types` - `dist/` — `tsup` build output (do not edit) @@ -14,7 +25,10 @@ - `tests/*.test.ts` — units plus `integration.test.ts`, which runs the whole loop over real sockets against a scripted OpenAI-compatible endpoint, a real - MCP server and a real authorization server (`tests/helpers/`). No key, ~5s. + MCP server and a real authorization server (`tests/helpers/`), and + `features-integration.test.ts`, which drives thinking, limits, compaction, + skills, approval, tool search and subagents (worker + in-process) the same + way. No key, ~7s. - `tests/live-model.test.ts` — the same loop against a REAL model. Skipped unless `AGENT_LIVE_MODEL_URL` points at an OpenAI-compatible endpoint; CI starts one via `live-model.yml`. Asserts mechanics only (the loop finished, a @@ -41,6 +55,9 @@ If `npm install` is needed (e.g. lockfile changed), run it with `--no-audit --no - Use `.ts` extensions in relative imports (project relies on `--experimental-strip-types`). - Zod v4 is used; when passing heterogeneous schemas through a shared array/iterable, type the collection as `z.ZodType` to avoid union-narrowing errors. - Anthropic's native structured output rejects `maxItems` on arrays — never add `.max()` to Zod arrays that flow into structured output. The guard test in `tests/anthropic-schema-compat.test.ts` enforces this. +- Keep system prompts run-stable (prompt caching): anything that changes per call (history, request, plan, trace) goes in the user prompt. Tests match stages on the role phrases "You are the Planner / Executor / Replanner / Synthesizer" — keep them. +- Tools reach the run through `runContext` (AsyncLocalStorage): emit, current step and the per-run state. Gate logic belongs inside `execute` (see `tool-wrap.ts`), not in the executor's stream loop. +- The public API, config fields, event names and semantics are shared with the browser sibling `@dudko.dev/agent-web`; keep names aligned when changing them. ## Boundaries diff --git a/README.md b/README.md index 91b7a9d..aa2577c 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,19 @@ The agent runs a plan → execute → replan → synthesize loop: 3. **Replan** — after each step the replanner decides whether to continue, revise the plan, or finish. 4. **Synthesize** — once finished, the synthesizer LLM writes the final answer for the user. -Multi-provider out of the box: OpenAI, Anthropic, Google (Gemini), xAI (Grok), Azure OpenAI, Amazon Bedrock, Google Vertex, DeepSeek, Vercel AI Gateway, Cloudflare Workers AI, and any OpenAI-compatible endpoint. Streaming events, per-run cancellation via `AbortSignal`, token budgets, retry/timeout, and concurrent runs on a single agent instance. +Multi-provider out of the box: OpenAI, Anthropic, Google (Gemini), xAI (Grok), Azure OpenAI, Amazon Bedrock, Google Vertex, DeepSeek, Vercel AI Gateway, Cloudflare Workers AI, and any OpenAI-compatible endpoint. Streaming events, per-run cancellation via `AbortSignal`, retry/timeout, and concurrent runs on a single agent instance. + +On top of the loop: + +- **[Thinking](#thinking-reasoning)** per stage, with the model's thoughts streamed as events. +- **[Token limits](#token-limits-and-step-caps)** — input / output / reasoning / total caps per run, enforced at every LLM step boundary, plus per-call output caps and a tool-call cap. +- **[Compaction](#compaction)** of long histories and traces (automatic and manual), and a cap on what the model sees of each tool result. +- **[Skills](#skills)** — agentskills.io-style `SKILL.md` instructions the planner picks and the executor loads on demand. +- **[Tool approval](#tool-approval-and-autopilot)** — autopilot, ask-for-writes, ask-for-all or read-only, with glob rules and a runtime toggle. +- **[Large MCP catalogs](#large-mcp-catalogs)** — tool search for hundreds of tools, `tools/list` pagination, connect timeouts. +- **[Prompt caching](#prompt-caching)** — run-stable system prompts, Anthropic cache breakpoints, OpenAI cache keys. +- **[Subagents](#subagents-worker_threads)** — a whole agent exposed as one tool, isolated in a `worker_thread`. +- **[Autonomy](#autonomy)** — the agent looks things up and decides on its own instead of asking. ## Install @@ -80,7 +92,18 @@ await agent.close() ### Streaming events -Pass an event handler as the second argument to `createAgent`, or per run via `onEvent`. Events include plan deltas, step starts and tool calls, replanner decisions, retries, budget breaches, the streamed final answer, and errors. See [`AgentEvent`](./src/types.ts) for the full union. +Pass an event handler as the second argument to `createAgent`, or per run via `onEvent`. Events include plan deltas, step starts and tool calls, replanner decisions, retries, budget breaches, the streamed final answer, and errors. See [`AgentEvent`](./src/types.ts) for the full union; the newer ones: + +| Event | When | +| --- | --- | +| `step.reasoning-delta` / `final.reasoning-delta` | The executor's / synthesizer's thoughts, when [thinking](#thinking-reasoning) is on and the provider streams them. | +| `budget.exceeded` | A run cap was crossed; `kind: 'input' \| 'output' \| 'reasoning' \| 'total' \| 'tool-calls'`, `tokens`, `cap`. | +| `context.compacted` | History or trace was compacted (`scope`, `beforeTokens`, `afterTokens`). | +| `skill.activated` | A skill became active (`by: 'plan' \| 'tool'`). | +| `tools.discovered` | `find_tools` activated tools (`query`, `names`). | +| `tool.approval-requested` / `tool.approval-resolved` | The approval gate asked / decided (`automatic` for rule, mode and timeout denials). | +| `subagent.start` / `subagent.event` / `subagent.complete` / `subagent.error` | A subagent tool call's lifecycle and its forwarded child events. | +| `usage` | Per LLM call; `phase` is `'plan' \| 'execute' \| 'replan' \| 'synthesize' \| 'compact' \| 'subagent'`. | ```ts const agent = await createAgent(config, (event) => { @@ -135,7 +158,7 @@ Caveats: - **Idempotency.** Resume re-enters the loop at the saved checkpoint (the iteration boundary after the last successful step). A crash mid-step means the in-flight step's tool calls are lost; on resume, the runner re-executes that step from scratch. If your tools have side effects (writes, payments, emails), the same call may fire twice. Design tools to be idempotent or guard against replay. - **Sandbox.** The per-run sandbox directory is auto-cleaned on completion. To resume, set `keepSandbox: true` so files written by earlier steps survive the crash. Without it, the trace's file references will point to a directory that no longer exists. - **Plan changes.** The saved `currentPlan` (post any revise) is what gets used on resume — the planner is not re-invoked. -- **Caps inherit.** `iterations` and `revisions` carry over, so per-run caps still apply across the boundary. +- **Caps inherit.** `iterations`, `revisions`, usage and the tool-call count (`toolCallCount`) carry over, so per-run caps still apply across the boundary; so do the active skills (`activeSkills`) and the trace summary (`traceSummary` / `traceSummaryUpTo`). - **Terminal status.** Resuming a run with `status: 'complete'` throws — re-read the saved `text` directly instead. - **Event semantics.** On resume the runner re-emits a `plan.created` event so consumers attaching mid-resume see the canonical plan; consumers that store every event will observe `plan.created` twice for the same `runId`. `onRunStart` is **not** re-fired (the original run already emitted it) — code that counts run starts must use `runId` for de-duplication. `onStepComplete` only fires for steps the resumed run actually executes; pre-resume steps are already in the loaded `trace`. - **Inputs are locked to the snapshot.** Both `options.input` and `options.history` passed to `agent.run({ resumeFromRunId })` are **silently ignored** in favor of the values stored at the original run start. This keeps the resumed prompt deterministic against the saved trace; pass an empty input (`input: ''`) to make the override explicit. @@ -186,6 +209,214 @@ await agent.reconnect() Nothing is printed or opened on your behalf; `onAuthorizationUrl` and the loopback listener are yours to wire. `agent.listTools()` stays empty until the flow completes — a server that needs authorization reports `needsAuthorization: true` in its connect result rather than a generic failure. `MCP_SERVERS` (the CLI's env config) is JSON, so it can only express `headers`; OAuth is a library-level feature. +## Thinking (reasoning) + +Turn on the model's reasoning for every stage with `thinking`, and override it per stage with `stageThinking` (a stage entry — even `false` — wins over the top-level value): + +```ts +const agent = await createAgent({ + ...config, + thinking: 'medium', // boolean | level | { level?, budgetTokens?, includeThoughts? } + stageThinking: { + planner: 'high', // plans and revisions deserve the most + executor: { level: 'low' }, + synthesizer: false, // provider default + }, +}) +``` + +| Setting | Sent to the provider | +| --- | --- | +| `false` / unset | nothing — the provider's default | +| `true` | `{ level: 'medium' }` | +| `'provider-default' \| 'minimal' \| 'low' \| 'medium' \| 'high' \| 'xhigh'` | the AI SDK's portable `reasoning` option, mapped onto each provider's own knob | +| `'none'` | `reasoning: 'none'` — explicitly off | +| `{ budgetTokens }` | an exact budget: Anthropic `thinking.budgetTokens`, Google `thinkingConfig.thinkingBudget` | +| `includeThoughts` (default `true`) | asks Google (`includeThoughts`) and OpenAI (`reasoningSummary: 'auto'`) to return their thoughts | + +`resolveThinking(setting)` is exported (pure) if you want to see exactly what a setting becomes. Provider-specific options always win over the portable level inside the SDK. + +The executor streams thoughts as `step.reasoning-delta` and the synthesizer as `final.reasoning-delta`. The planner and replanner use structured output, which has no reasoning stream. Compaction calls never think. Usage reports `reasoningTokens` (part of `outputTokens`). + +> Some models refuse thinking together with forced structured output. If the planner fails with thinking on, disable it there: `stageThinking: { planner: false, replanner: false }`. + +## Token limits and step caps + +```ts +const agent = await createAgent({ + ...config, + limits: { + maxInputTokens: 400_000, // cumulative per run + maxOutputTokens: 40_000, // includes reasoning + maxReasoningTokens: 20_000, + maxTotalTokens: 500_000, // input + output (falls back to the legacy top-level maxTotalTokens) + perCall: { planner: 2_000, executor: 4_000, replanner: 1_000, synthesizer: 3_000, compaction: 1_024 }, + }, + maxToolCalls: 40, // tool calls per run, across steps + maxPlanSteps: 6, // hard cap on plan steps (default 8) +}) +``` + +The run caps are **soft**: they are checked between steps *and* inside an executor step, as an extra `stopWhen` condition that adds the usage of the step's LLM calls to the run so far — so a runaway tool loop stops at its next LLM step instead of at the end of the step. When a cap is crossed the agent emits `budget.exceeded` (`kind: 'input' | 'output' | 'reasoning' | 'total' | 'tool-calls'`), stops executing steps and goes straight to synthesis, which still runs (capped by `perCall.synthesizer`). `checkLimits(usage, limits)` is the pure check, exported. + +`maxToolCalls`: once reached, further tool calls in the run fail with `tool-call budget exhausted` (the model sees that as the tool result) and the run moves to synthesis with `budget.exceeded` of kind `'tool-calls'`. Built-in tools (`find_tools`, `load_skill`, `read_skill_file`) do not count. + +The other step caps are unchanged: `maxIterations` (executed steps per run), `maxStepsPerTask` (LLM steps inside one executor call) and `maxRevisions` (replanner revisions per run). + +`IUsage` reports `reasoningTokens`, `cachedInputTokens` (served from the provider's prompt cache) and `cacheWriteTokens` alongside the totals; `normalizeUsage` / `addUsage` are exported. + +## Compaction + +Long conversations and long runs are compacted automatically; everything is on by default and tuned with `compaction`: + +```ts +compaction: { + auto: true, // default + contextWindowTokens: 128_000, // default + thresholdTokens: 64_000, // default: 50% of the window + keepRecentTurns: 4, // history turns kept verbatim + keepRecentSteps: 3, // trace steps kept verbatim + summaryMaxTokens: 1024, + maxToolOutputChars: 20_000, // per tool result the MODEL sees; 0 = unlimited + clearToolResultsAfterTokens: 32_000, // default: 25% of the window; 0 = never + keepToolResults: 3, // newest tool results kept verbatim when clearing +} +``` + +- **History, at run start.** When the estimated history (`estimateTokens` = chars / 4) exceeds the threshold, all but the last `keepRecentTurns` turns become ONE assistant turn starting with `[Summary of earlier conversation]`. The run works on the compacted copy; your array is never mutated, and the result carries it as `result.compactedHistory` so you can store it instead. +- **Trace, before each executor / replanner / synthesizer call.** Over the threshold, every step but the last `keepRecentSteps` is folded into a running summary; prompts show `Summary of earlier steps: …` followed by the recent steps verbatim. The trace itself stays intact in the result, the events and persistence; the summary is saved in the snapshot (`traceSummary`, `traceSummaryUpTo`) so a resumed run keeps it. +- **Tool results.** Every tool (MCP, native, built-in) gets a `toModelOutput` that clips what the model sees to `maxToolOutputChars` with a `… [truncated N chars]` marker. A tool's own `toModelOutput` runs first. The raw output still lands in the trace and in `step.tool-result`. +- **Stale tool results inside a step** (context editing, like Anthropic's `clear_tool_uses` and Claude Code's micro-compaction). Once a step's tool loop grows past `clearToolResultsAfterTokens`, the oldest results are replaced with a one-line stub (`[ result cleared to save context — call the tool again if you still need it]`), keeping the newest `keepToolResults`. The calls stay, so the model knows what it did. Clearing is sticky (the cached prefix doesn't flip back) and keyed by position, so servers that repeat tool-call ids are fine. `createToolResultClearer` is exported. + +Each compaction emits `context.compacted` (`scope: 'history' | 'trace' | 'tool-results'`, `beforeTokens`, `afterTokens`); the summary call's usage is reported as `usage` with `phase: 'compact'` and counts toward the run. The synthesizer model writes the summaries. + +Manual: `agent.compact({ history, force?, signal? })` returns `{ history, summary?, beforeTokens, afterTokens, compacted }`; the CLI's `/compact` does exactly that for the REPL history. `compactHistory(turns, model, opts)` is exported for use without an agent. Compaction never throws — on any failure the input comes back unchanged. + +## Skills + +Skills are packaged instructions in the [agentskills.io](https://agentskills.io) format: a folder per skill with a `SKILL.md` (YAML frontmatter with `name` and `description`, then the instructions) and any bundled text files. + +```ts +import { createAgent, loadSkillsFromDir, parseSkillMarkdown, defineSkill } from '@dudko.dev/agent' + +const skills = await loadSkillsFromDir('./skills') // every ./skills//SKILL.md +const agent = await createAgent({ ...config, skills }) +agent.listSkills() // [{ name, description }] +``` + +Progressive disclosure keeps them cheap: + +1. The planner and executor system prompts carry only an index: `- name: description`. +2. The planner's plan has an optional `skills` field naming the skills that apply (unknown names are dropped with a warning). Their instructions are injected into the executor, replanner and synthesizer system prompts for the rest of the run (24k chars in total, clipped beyond that). +3. The executor gets two read-only built-in tools: `load_skill({ name })` returns `{ name, content, files }` and activates the skill for later steps; `read_skill_file({ name, path })` returns a bundled file. Both always run without approval. + +Every activation emits `skill.activated` (`by: 'plan' | 'tool'`). A host or MCP tool named `load_skill` / `read_skill_file` is a configuration error. `parseSkillMarkdown(markdown, files?)` and `defineSkill(skill)` validate skills you build yourself (names match `[a-z0-9-]{1,64}`). `loadSkillsFromDir` bundles text files (`.md .txt .json .yaml .yml .csv .ts .js .py .sh .html .css .xml .toml`, up to 256 KB each). + +## Tool approval and autopilot + +By default the agent runs every tool call (**autopilot**). `toolApproval` puts the host in charge of consent: + +```ts +const agent = await createAgent({ + ...config, + toolApproval: { + mode: 'ask-writes', // 'autopilot' | 'ask-writes' | 'ask-all' | 'read-only' + rules: { 'github__delete_*': 'deny', github__get_me: 'allow' }, + onRequest: async ({ toolName, input, readOnly, step, runId }) => { + const ok = await askTheUser(toolName, input) + return { approved: ok, remember: ok } // or just true / false + }, + timeoutMs: 60_000, // no answer -> deny + }, +}) + +agent.setToolApprovalMode('autopilot') // the toggle; applies to the next call, in-flight runs included +agent.getToolApprovalMode() +``` + +Each call is decided in this order: (1) a tool approved earlier with `remember: true` runs; (2) the most specific matching rule — exact name over the longest `*` glob — says `allow`, `ask` or `deny`; (3) the mode: `autopilot` allows, `read-only` allows read-only tools and denies the rest, `ask-writes` allows read-only tools and asks for the rest, `ask-all` asks. Built-in tools are always allowed. An `ask` without `onRequest` is a denial. + +A tool is read-only when its MCP descriptor says `annotations.readOnlyHint: true`, or when a native tool is marked with `markReadOnly(tool)` (`isReadOnlyTool(tool)` checks). The request shows the input after `inputSanitizer`. + +A denied call throws `ToolDeniedError` inside the tool, so the step records `ok: false` and the model reads *"Tool call denied by the user… Do not retry it; continue without it, or report what is blocked."* — the replanner sees it as any failed call. Events: `tool.approval-requested` (only when actually asking) and `tool.approval-resolved` (`automatic: true` for rule / mode / timeout denials; automatic allows emit nothing). The gate runs inside every tool's `execute` — MCP, native and subagent tools alike. + +## Large MCP catalogs + +**Tool search.** `toolSelectionStrategy` now defaults to `'auto'`: the executor gets every tool (`'all'`, exactly as before) while the filtered catalogue has at most `toolSearchThreshold` (default 40) tools, and switches to `'search'` above that. In `'search'` mode each step starts with the built-in tools, the step's `suggestedTools` and the tools discovered earlier in the run (the 16 most recent), plus `find_tools({ query, server?, limit? })`. It ranks the whole catalogue and **activates** the matches: they become callable from the model's next LLM step and stay active for the rest of the run (`tools.discovered` event). The planner sees an abridged catalogue grouped by server and treats `suggestedTools` as hints. The ranking is the exported pure `searchTools(catalog, query, { server?, limit? })` — terms matched in the name score 3, in the server 2, in the description 1, prefix-matched for terms of 3+ chars. `'all'` and `'plan-narrowed'` behave as before. + +**Loading.** `tools/list` follows `nextCursor` (up to 100 pages), on connect and on `tools/list_changed` refreshes. Each server must connect and list within `connectTimeoutMs` (per server; default from the top-level `mcpConnectTimeoutMs`, 30 s; `0` disables) or it is reported failed (`connect timed out after Xms`) and closed while the others mount. `agent.listTools()` entries carry `server` and `readOnly`. + +## Prompt caching + +`promptCaching` is on by default. Every stage keeps its **system** prompt run-stable — role instructions, domain context (`systemPrompt`), the skills index, the tool catalogue where the stage shows one, and the active skills — and puts everything that changes (history, request, plan, trace) in the user message. That alone lets OpenAI and Gemini implicit caches hit. On top of it the system prompt is sent as a system message with an Anthropic ephemeral cache breakpoint, and each call carries an OpenAI `promptCacheKey` (`${clientName}:${stage}`). Unknown provider keys are ignored by other providers. + +```ts +promptCaching: { ttl: '1h', key: 'my-app' } // or false to send plain system strings +``` + +Inside a step's tool loop a second Anthropic breakpoint **rolls to the newest message** every round, so each round reads all earlier rounds from the cache and writes only the new tail (`withRollingBreakpoint`; earlier message breakpoints are removed, so a request carries at most two — Anthropic allows four). Tools are sent and catalogued **sorted by name**, so the tool list that heads every cached prefix is identical however the MCP servers connected. + +Usage reports `cachedInputTokens` / `cacheWriteTokens`, so the effect is measurable. + +### Token efficiency, in one place + +The practices coding agents (Claude Code, the Claude and GitHub Copilot extensions) use to keep long sessions cheap, and the knob for each: + +| Practice | How | Knob | +| --- | --- | --- | +| Stable, cacheable prefix | run-stable system prompts, sorted tools, dynamic content last | `promptCaching` | +| Cache the growing loop | rolling Anthropic breakpoint; OpenAI `promptCacheKey` per stage | `promptCaching: { ttl }` | +| Don't send every tool | compact catalogue + `find_tools` above the threshold (deferred tools) | `toolSelectionStrategy`, `toolSearchThreshold` | +| Load instructions on demand | skills: name + description in the prompt, body on activation | `skills` | +| Cap tool output | per result, as the model sees it | `compaction.maxToolOutputChars` | +| Clear stale tool results | oldest results in a long loop become stubs | `compaction.clearToolResultsAfterTokens` | +| Auto-compact | history and trace summarised past a threshold; `agent.compact()` | `compaction` | +| Isolate side quests | subagents (worker_threads) with their own context; only the answer returns | `createSubagentTool` | +| Pass results, not transcripts | later steps and the answer get step summaries + clipped findings | — | +| Bound everything | tokens per run / kind, per-call output caps, tool-call and plan-step caps | `limits`, `maxToolCalls`, `maxPlanSteps` | +| Think only where it pays | thinking per stage | `stageThinking` | +| Cheaper model for summaries | compaction and the answer use the synthesizer model | `synthesizer` stage | + +## Subagents (worker_threads) + +`createSubagentTool` exposes a whole agent as one tool of a parent agent: input `{ task }`, output the child's final text. + +```ts +import { createAgent, createSubagentTool } from '@dudko.dev/agent' + +const researcher = createSubagentTool({ + name: 'researcher', + description: 'Research a self-contained question and report the findings', + config: { ...childConfig }, // a full IAgentConfig for the child + isolation: 'worker', // default; or 'in-process' + tools: { lookup }, // host tools for the child + maxConcurrent: 4, // parallel calls beyond this queue + timeoutMs: 120_000, + outputMaxChars: 8_000, + readOnly: true, // for tool approval +}) + +const agent = await createAgent({ ...config, tools: { researcher } }) +``` + +- **`'worker'`** runs each call in its own `worker_thread` (terminated afterwards), so the child's CPU work and crashes stay off the parent's event loop. Its config must be structured-cloneable — a function anywhere (`getHeaders`, `authProvider`, `fetch`, sanitizers, `persistence`, `onRequest`, native `tools`) throws, naming the field. Host tools passed as `tools` are **proxied**: the child sees their schemas, every call runs on the parent thread. +- **`'in-process'`** builds the child on this thread; `tools` are merged into its config. + +Several calls in one model step run in parallel (bounded by `maxConcurrent`). The parent's abort aborts the child. The child's usage is emitted on the parent as `usage` with `phase: 'subagent'` as it happens, so it counts against the parent's limits. Parent events: `subagent.start`, `subagent.event` (each child event, long strings clipped), `subagent.complete` (`text`, `usage`), `subagent.error`. + +The worker entry ships as `dist/subagent-worker.js` and is resolved next to the bundle, from the ESM and the CJS build alike. + +## Autonomy + +The stage prompts are written for an agent that acts on its own: + +- The **planner** plans tool use whenever the tools can obtain what the request needs — a question that needs data is actionable — and never plans a step that asks the user. An ambiguous request gets the most reasonable interpretation, with the assumption written into the step. +- The **executor** never asks questions or for confirmation, looks things up with tools, picks sensible defaults and states its assumptions in the step result. It ends with `[BLOCKER]` only when a step is truly impossible (missing credentials or permissions, a denied tool, unavailable data). +- The **replanner** prefers revising around a failure over finishing with nothing, and never revises into an "ask the user" step. +- The **synthesizer** reports what was done and found, mentions assumptions, and ends with a question only when the request genuinely cannot be completed without the user. + +Consent is the host's job ([tool approval](#tool-approval-and-autopilot)) — the model never asks for permission in text. + ## Configuration `createAgent(config)` accepts an [`IAgentConfig`](./src/types.ts). Highlights: @@ -199,16 +430,26 @@ Nothing is printed or opened on your behalf; `onAuthorizationUrl` and the loopba | `model` | Default model for every stage (executor / planner / synthesizer) when no per-stage override is set. | | `planner` / `synthesizer` | Optional per-stage override blocks: `{ providerType?, baseURL?, apiKey?, model? }`. Use these to mix providers (e.g. Gemini planner, Anthropic synthesizer). Cross-provider overrides MUST set their own `apiKey`. | | `plannerModel` / `synthesizerModel` | **Deprecated** model-only shortcuts. Equivalent to `planner: { model }` / `synthesizer: { model }`. The override block, if present, wins. | -| `mcpServers` | `Record` — StreamableHTTP for remote (legacy HTTP+SSE servers are **not** supported), stdio for locally-spawned servers. | -| `tools` | Optional `ToolSet` of native AI-SDK tools registered alongside MCP-discovered ones. Names must not collide with MCP-prefixed names (`createAgent` throws on conflict). | +| `mcpServers` | `Record` — StreamableHTTP for remote (legacy HTTP+SSE servers are **not** supported), stdio for locally-spawned servers. | +| `tools` | Optional `ToolSet` of native AI-SDK tools registered alongside MCP-discovered ones. Names must not collide with MCP-prefixed names or the built-in tools (`createAgent` throws on conflict). Mark read-only tools with `markReadOnly(tool)`. | | `availableTools` / `excludedTools` | Whitelist / blacklist applied to **all** tools (MCP and native). | | `maxIterations` | Cap on **executed steps** across the run (every step counts, including those run after a `revise`). | | `maxStepsPerTask` | Cap on LLM steps inside a single executor call (multi-step tool calling). | | `maxRevisions` | Cap on `revise` decisions the replanner can make per run. Default `2`. | | `replanAfter` | Replan trigger: `'failure'` (default; blocked step or a tool failure that stayed failed) \| `'always'` \| `(stepResult) => boolean \| Promise` (bounded by `llmTimeoutMs`; falls back to `'failure'` on error). | -| `maxTotalTokens` | Soft cap on cumulative input + output tokens; checked between steps and triggers an early jump to synthesis when crossed. | +| `maxTotalTokens` | Soft cap on cumulative input + output tokens; checked between steps and triggers an early jump to synthesis when crossed. Legacy shortcut for `limits.maxTotalTokens` (which wins). | +| `limits` | `{ maxInputTokens?, maxOutputTokens?, maxReasoningTokens?, maxTotalTokens?, perCall?: { planner?, executor?, replanner?, synthesizer?, compaction? } }` — cumulative per-run caps (also enforced inside an executor step) and per-call output caps. See [Token limits](#token-limits-and-step-caps). | +| `maxToolCalls` | Cap on tool calls per run; further calls fail with "tool-call budget exhausted" and the run goes to synthesis. | +| `maxPlanSteps` | Hard cap on plan steps (initial and revised). Default `8`. | +| `thinking` / `stageThinking` | Reasoning for every stage / per stage (`planner`, `executor`, `replanner`, `synthesizer`). See [Thinking](#thinking-reasoning). | +| `compaction` | `{ auto?, contextWindowTokens?, thresholdTokens?, keepRecentTurns?, keepRecentSteps?, summaryMaxTokens?, maxToolOutputChars? }`. See [Compaction](#compaction). | +| `skills` | `ISkill[]` — see [Skills](#skills). | +| `toolApproval` | `{ mode?, rules?, onRequest?, timeoutMs? }` — default autopilot. See [Tool approval](#tool-approval-and-autopilot). | | `llmTimeoutMs` / `llmMaxRetries` | Per-LLM-call timeout and retry budget. | -| `toolSelectionStrategy` | `'all'` (default) gives the executor every tool each step; `'plan-narrowed'` exposes only `step.suggestedTools`. | +| `toolSelectionStrategy` | `'auto'` (default: `'all'` up to `toolSearchThreshold` tools, `'search'` above), `'all'` (every tool each step), `'plan-narrowed'` (only `step.suggestedTools`), `'search'` (built-ins + suggested + discovered, plus `find_tools`). | +| `toolSearchThreshold` | Catalogue size above which `'auto'` switches to `'search'`. Default `40`. | +| `mcpConnectTimeoutMs` | Connect + `tools/list` budget per MCP server (each server's `connectTimeoutMs` wins). Default `30000`; `0` disables. | +| `promptCaching` | `true` (default) \| `false` \| `{ ttl?: '5m' \| '1h', key? }`. See [Prompt caching](#prompt-caching). | | `outputSanitizer` | Optional `(toolName, output) => unknown` hook to redact tool results before they reach the LLM. | | `inputSanitizer` | Optional `(toolName, input) => unknown` hook to redact LLM-generated tool args before they hit the MCP server **and** before they appear in `step.tool-call` events. **Must be idempotent** — applied at both the event boundary and the dispatch boundary. | | `outputSanitizer` ordering | The sanitizer runs on the **raw MCP `result.content`** (image/audio base64 still inline), **before** the agent spills binary parts to the sandbox. This favors privacy: a sanitizer that drops a sensitive image keeps the bytes out of the disk entirely. If you want post-spill sanitization (e.g. redact a path), do it in your tool wrapper instead. | @@ -247,13 +488,25 @@ Notes: ```ts interface IAgent { run(options: IAgentRunOptions): Promise - listTools(): { name: string; description: string }[] + listTools(): { name: string; description: string; server?: string; readOnly?: boolean }[] + listSkills(): { name: string; description: string }[] + compact(options: { history: IConversationTurn[]; signal?: AbortSignal; force?: boolean }): Promise<{ + history: IConversationTurn[] + summary?: string + beforeTokens: number + afterTokens: number + compacted: boolean + }> + setToolApprovalMode(mode: ToolApprovalMode): void + getToolApprovalMode(): ToolApprovalMode reconnect(): Promise close(options?: { waitForRuns?: boolean; timeoutMs?: number }): Promise activeRuns(): number } ``` +`IAgentRunResult` is `{ text, plan, trace, iterations, usage, compactedHistory? }`. + A single agent instance supports concurrent `run()` calls — each gets its own `runId` (via `AsyncLocalStorage`), usage accumulator, abort signal, and `onEvent`. Tools and models are shared. Top-level exports beyond `createAgent`: @@ -261,6 +514,14 @@ Top-level exports beyond `createAgent`: - `getCurrentRunId(): string | undefined` — read the active run's id from any code reachable from `agent.run()` (planner, executor, MCP `execute`, retry sleeps, …). Useful for correlating logs/metrics across concurrent runs on a single agent instance. - `getCurrentRunSandbox(): string | undefined` — absolute path to the active run's sandbox directory. Native tools that need to spill binary output should write into this path so files are auto-cleaned when the run completes (set `keepSandbox: true` to retain). - `redactHeaders(headers)` — small helper for masking `Authorization`, `X-Api-Key`, `Cookie`, etc. when logging request headers (e.g. inside an `outputSanitizer` or your own MCP transport wrapper). +- `createSubagentTool(options)` — [subagents](#subagents-worker_threads). +- `resolveThinking(setting)`, `mergeProviderOptions(...layers)` — [thinking](#thinking-reasoning). +- `checkLimits(usage, limits)`, `resolveLimits(config)`, `normalizeUsage(sdkUsage)`, `addUsage(a, b)` — [token limits](#token-limits-and-step-caps). +- `compactHistory(turns, model, opts)`, `estimateTokens(text)`, `withToolOutputLimit(tool, maxChars)`, `truncateToolModelOutput(output, maxChars)` — [compaction](#compaction). +- `parseSkillMarkdown(markdown, files?)`, `defineSkill(skill)`, `loadSkillsFromDir(dir)` — [skills](#skills). +- `markReadOnly(tool)`, `isReadOnlyTool(tool)`, `decideToolPermission(input)`, `matchToolRule(rules, name)`, `ToolDeniedError`, `ToolBudgetError` — [tool approval](#tool-approval-and-autopilot). +- `searchTools(catalog, query, opts)` — [tool search](#large-mcp-catalogs). +- `resolvePromptCaching(setting)`, `buildInstructions(system, caching)` — [prompt caching](#prompt-caching). ### Closing the agent @@ -278,7 +539,9 @@ The package ships a REPL CLI as `dd-agent`. After install, npm makes it availabl dd-agent --env-file=.env ``` -`--env-file=` is loaded via Node's built-in `process.loadEnvFile`, so no `dotenv` dependency is needed. Without the flag the CLI reads the ambient process env. `-h` / `--help` prints the supported flags and the in-REPL slash commands (`/status`, `/tools`, `/history`, `/reset`, `/reconnect`, `/exit`). +`--env-file=` is loaded via Node's built-in `process.loadEnvFile`, so no `dotenv` dependency is needed. Without the flag the CLI reads the ambient process env. `-h` / `--help` prints the supported flags and the in-REPL slash commands (`/status`, `/tools`, `/history`, `/reset`, `/compact`, `/autopilot`, `/approval `, `/reconnect`, `/exit`). + +The newer features are env-driven too: `AGENT_THINKING` (`off | minimal | low | medium | high | xhigh | `), `AGENT_MAX_INPUT_TOKENS` / `AGENT_MAX_OUTPUT_TOKENS` / `AGENT_MAX_REASONING_TOKENS` / `AGENT_MAX_TOTAL_TOKENS`, `AGENT_MAX_TOOL_CALLS`, `AGENT_TOOL_APPROVAL` (`autopilot | ask-writes | ask-all | read-only`), `AGENT_SKILLS_DIR`, `AGENT_TOOL_SELECTION_STRATEGY` (`auto | all | plan-narrowed | search`), `AGENT_CONTEXT_WINDOW_TOKENS` and `AGENT_COMPACTION=off`. The REPL prints the model's thoughts dimmed, auto-compacts its history with the same settings, and — when the approval mode asks — prompts `allow? [y]es / [n]o / [a]lways` right in the terminal; `/autopilot` toggles between autopilot and the last asking mode. For local development against the source tree: @@ -325,6 +588,8 @@ parsed — never the wording, which would make it a coin flip. - **MCP connect failures.** By default `createAgent` is fail-tolerant: a server that can't connect is logged at `error` level and skipped. The agent still starts with whatever tools did mount. Set `failOnNoTools: true` to throw when **every** configured server failed. Servers connect concurrently, so one slow server no longer delays the ones declared after it; tools still mount in declaration order. - **MCP tool errors.** A tool result carrying `isError: true` is surfaced as a **thrown** tool error, so the step records `ok: false` and `replanAfter: 'failure'` sees it. The error text the server returned becomes the error message (after `outputSanitizer`, if you set one). - **MCP tool names.** Server and tool names are sanitized into `[a-zA-Z0-9_-]` (what OpenAI, Anthropic and Gemini accept) before being joined as `server__tool`, and a collision produced by that mapping gets a `_2` suffix rather than being dropped. `callTool` always uses the server's original name. If you pin `availableTools` / `excludedTools`, use the sanitized names. +- **Tool-output cap.** Since compaction landed, the model sees at most `compaction.maxToolOutputChars` (default 20 000) of each tool result; set it to `0` for the old unlimited behaviour. The trace and events keep the full output. +- **Default tool strategy.** `toolSelectionStrategy` defaults to `'auto'`, which is `'all'` (the previous default) up to 40 tools; set `'all'` explicitly to keep every tool in every step regardless of catalogue size. - **Blocker detection.** When the executor cannot complete a step it ends its reply with the literal `[BLOCKER]` token; the agent strips the token from the surfaced summary and sets `IStepResult.blocked = true`, which triggers the replanner. The detection is structural and language-independent — works regardless of the language the executor wrote in. - **Retry duplicates in events.** Executor LLM retries (5xx / 429 / network) restart `streamText`, so consumers may observe `step.text-delta` / `step.tool-call` / `step.tool-result` events repeated for the same step. The `retry` event with `phase: 'execute'` precedes each repeat — UIs should clear any per-step buffers on it. - **Mid-stream thought rewrites.** Some providers (notably Gemini structured outputs) rewrite `partialObjectStream.thought` from scratch instead of appending. The agent emits a single `log`-level warning and stops streaming `plan.thought-delta` for that run; the canonical thought still arrives in `plan.created`. diff --git a/env.example b/env.example index 8d6e883..3bded8d 100644 --- a/env.example +++ b/env.example @@ -105,19 +105,57 @@ AGENT_MAX_STEPS_PER_TASK=8 # Prevents the LLM from looping in "revise -> execute -> revise". Default 2. # AGENT_MAX_REVISIONS=2 -# Hard cap on total tokens per run (input + output). When crossed the agent -# exits the execution loop and proceeds to synthesize the final answer. -# AGENT_MAX_TOTAL_TOKENS=200000 +# Per-run token caps. When one is crossed (checked between steps AND at every +# LLM step inside an executor step) the agent stops executing steps and +# synthesizes the final answer from what it has. +# AGENT_MAX_TOTAL_TOKENS=200000 # input + output +# AGENT_MAX_INPUT_TOKENS=150000 +# AGENT_MAX_OUTPUT_TOKENS=30000 # includes reasoning +# AGENT_MAX_REASONING_TOKENS=20000 + +# Cap on tool calls per run (across steps). Further calls fail with +# "tool-call budget exhausted" and the run goes to the final answer. +# AGENT_MAX_TOOL_CALLS=40 + +# Thinking / reasoning for every stage: +# off explicitly disabled +# minimal|low|medium|high|xhigh portable effort level +# provider-default the provider's default level +# an exact budget in tokens (Anthropic / Google) +# Unset: nothing is sent (the provider's default behaviour). +# AGENT_THINKING=medium + +# Tool approval (consent) mode: +# autopilot (default) every tool call runs +# ask-writes read-only tools run, others ask at the REPL prompt +# ask-all every call asks +# read-only read-only tools run, others are denied +# Answer y / n / a(lways) at the prompt; /autopilot toggles at runtime and +# /approval switches modes. +# AGENT_TOOL_APPROVAL=ask-writes + +# Folder of agentskills.io-style skills: //SKILL.md (+ bundled +# text files). The planner picks the ones that apply; the executor can load +# any with load_skill. +# AGENT_SKILLS_DIR=./skills + +# Compaction: history / trace summaries kick in at 50% of the context window. +# AGENT_COMPACTION=off disables the automatic part (/compact still works). +# AGENT_CONTEXT_WINDOW_TOKENS=128000 +# AGENT_COMPACTION=on # Tool selection strategy for the executor: -# - all (default) every step sees the full filtered ToolSet. -# Works well with <=50 tools. +# - auto (default) 'all' up to 40 tools, 'search' above. +# - all every step sees the full filtered ToolSet. +# Works well with <=40 tools. # - plan-narrowed executor gets only the tools listed in # step.suggestedTools. The planner MUST populate # suggestedTools for every step that needs tools. -# Cuts context tokens significantly when the catalog -# has 100+ tools. -# AGENT_TOOL_SELECTION_STRATEGY=all +# - search each step starts with the suggested tools and the +# ones found earlier, plus find_tools, which searches +# the whole catalogue and activates matches. For +# catalogues of hundreds of tools. +# AGENT_TOOL_SELECTION_STRATEGY=auto # Fine-grained MCP tool control. Names use the `serverName__toolName` format. # availableTools wins over excludedTools. diff --git a/package-lock.json b/package-lock.json index d9d5901..b0831e7 100644 --- a/package-lock.json +++ b/package-lock.json @@ -27,24 +27,24 @@ ], "license": "MIT", "dependencies": { - "@modelcontextprotocol/sdk": "^1.30.1", + "@modelcontextprotocol/sdk": "^1.32.0", "@opentelemetry/api": "^1.9.1", - "ai": "^7.0.116", + "ai": "^7.0.127", "zod": "^4.6.5" }, "bin": { "dd-agent": "dist/cli.js" }, "devDependencies": { - "@ai-sdk/amazon-bedrock": "^5.0.96", - "@ai-sdk/anthropic": "^4.0.65", - "@ai-sdk/azure": "^4.0.81", - "@ai-sdk/deepseek": "^3.0.54", - "@ai-sdk/google": "^4.0.82", - "@ai-sdk/google-vertex": "^5.0.95", - "@ai-sdk/openai": "^4.0.77", - "@ai-sdk/openai-compatible": "^3.0.57", - "@ai-sdk/xai": "^5.0.10", + "@ai-sdk/amazon-bedrock": "^5.0.105", + "@ai-sdk/anthropic": "^4.0.71", + "@ai-sdk/azure": "^4.0.90", + "@ai-sdk/deepseek": "^3.0.58", + "@ai-sdk/google": "^4.0.87", + "@ai-sdk/google-vertex": "^5.0.101", + "@ai-sdk/openai": "^4.0.83", + "@ai-sdk/openai-compatible": "^3.0.62", + "@ai-sdk/xai": "^5.0.14", "@types/node": "^22.9.0", "prettier": "^3.9.9", "tsup": "^8.5.1", @@ -100,16 +100,16 @@ } }, "node_modules/@ai-sdk/amazon-bedrock": { - "version": "5.0.96", - "resolved": "https://registry.npmjs.org/@ai-sdk/amazon-bedrock/-/amazon-bedrock-5.0.96.tgz", - "integrity": "sha512-eh/orxOe7r/RFjMi0j1uJjeJQeL2X2v+V9dl84nBQfTpqAGsAAZht61uiFvDP+YbqjjnDjFgzCf3xFVicUIKgQ==", + "version": "5.0.105", + "resolved": "https://registry.npmjs.org/@ai-sdk/amazon-bedrock/-/amazon-bedrock-5.0.105.tgz", + "integrity": "sha512-IWO8ow0MRW5914VyOelqWUf+zjSkmUj3K+mArRn152oH3SnFjhleVWwFQ54RHJGsEf0aHfnVbd8wL6c7HevPyA==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/anthropic": "4.0.65", - "@ai-sdk/openai": "4.0.77", - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49", + "@ai-sdk/anthropic": "4.0.71", + "@ai-sdk/openai": "4.0.83", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53", "@smithy/eventstream-codec": "^4.3.3", "@smithy/util-utf8": "^4.3.3", "aws4fetch": "^1.0.20" @@ -122,14 +122,14 @@ } }, "node_modules/@ai-sdk/anthropic": { - "version": "4.0.65", - "resolved": "https://registry.npmjs.org/@ai-sdk/anthropic/-/anthropic-4.0.65.tgz", - "integrity": "sha512-3fmxxn5xWGOPU1wGrUCI30YFW5r6h2WrY+sz3rokkVlBHrDr889hPYP1yNoSmOVvtAwCFWlm79P2lRngIPOObg==", + "version": "4.0.71", + "resolved": "https://registry.npmjs.org/@ai-sdk/anthropic/-/anthropic-4.0.71.tgz", + "integrity": "sha512-wiv3jhUH0RrvGzSolJMZfZ2cI9As45LhwYmcAoWvNivXjC7wzJ4W/Udq9/u7EGnGsT8lpmFKEKBrMXVeUO97Dg==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -139,16 +139,16 @@ } }, "node_modules/@ai-sdk/azure": { - "version": "4.0.81", - "resolved": "https://registry.npmjs.org/@ai-sdk/azure/-/azure-4.0.81.tgz", - "integrity": "sha512-GRuumGejrXvV3c6KA/8b58HB6UgPnrkoxcWDJVeoL/fGs2xNtPO3WNWr596xTdn5IdYbbV1YRSJTu6rapLHCLA==", + "version": "4.0.90", + "resolved": "https://registry.npmjs.org/@ai-sdk/azure/-/azure-4.0.90.tgz", + "integrity": "sha512-11eOmyTsCcKASN1LMGds9p3GFydgRJRzCq9xKnioJ5o+DHd7sQ+FR6YUiiwIFWk8+Oj8kcS1Aldr/5JvZCY6SA==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/deepseek": "3.0.54", - "@ai-sdk/openai": "4.0.77", - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/deepseek": "3.0.58", + "@ai-sdk/openai": "4.0.83", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -158,14 +158,14 @@ } }, "node_modules/@ai-sdk/deepseek": { - "version": "3.0.54", - "resolved": "https://registry.npmjs.org/@ai-sdk/deepseek/-/deepseek-3.0.54.tgz", - "integrity": "sha512-1ByKYV/uis+3U66kueMJgF2DjN0DcJVtacZf7s1+BttKQwIRmHhKnQ3aqDRlIzpc2as59eqO5dxGi8Bfii2Pnw==", + "version": "3.0.58", + "resolved": "https://registry.npmjs.org/@ai-sdk/deepseek/-/deepseek-3.0.58.tgz", + "integrity": "sha512-tOafL69mHQGJH3g/mu4k0evkP/cRLSHCWe7L9rcVV4b0S1zI2rpWDZbToZO7D4X/uox9baVmgRQ5AoUTeaDs0A==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -175,13 +175,13 @@ } }, "node_modules/@ai-sdk/gateway": { - "version": "4.0.94", - "resolved": "https://registry.npmjs.org/@ai-sdk/gateway/-/gateway-4.0.94.tgz", - "integrity": "sha512-hcp7GLnMynH6NfMnemgvW9acn7j3WsUe/dHFzX3/fhPit08xhm1aYmDGHMiIT6YwsXVBxkit9TriKb2SGAOg+Q==", + "version": "4.0.103", + "resolved": "https://registry.npmjs.org/@ai-sdk/gateway/-/gateway-4.0.103.tgz", + "integrity": "sha512-nnGTUHzPRQ14jhv9CL4OQrzcRBIFTBMIPKGIi9Z9u/4GtwSRxa/nxUX6np2Iwp5phOzO0a0njj/j9Jolf5AEaA==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53", "@vercel/oidc": "3.2.0" }, "engines": { @@ -192,14 +192,14 @@ } }, "node_modules/@ai-sdk/google": { - "version": "4.0.82", - "resolved": "https://registry.npmjs.org/@ai-sdk/google/-/google-4.0.82.tgz", - "integrity": "sha512-2zPYyBSyIcywLtOEJRxlrtXjBjz4KxEU19Qqg/Dk8I9BrsB08+ghG2jv155dtuDZj8qZMrh6LO8J+Wt5O4xVhg==", + "version": "4.0.87", + "resolved": "https://registry.npmjs.org/@ai-sdk/google/-/google-4.0.87.tgz", + "integrity": "sha512-s0tWc+QeSce7O2ZLm+VWSRQy16dGfqEcMIUxEWh+LFvXX3h0TLH8jxCQ2H1qhiBgTta+6A97eHPhBlRFu1xy+g==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -209,17 +209,17 @@ } }, "node_modules/@ai-sdk/google-vertex": { - "version": "5.0.95", - "resolved": "https://registry.npmjs.org/@ai-sdk/google-vertex/-/google-vertex-5.0.95.tgz", - "integrity": "sha512-BIqJYwHu+348dvw8eQ1scY6Bqg4a3JE6TmJvGOHYlJfcVrzvSGRrzmA+6OicPLRQmSq9/mLp54pU9bmqYS+CZA==", + "version": "5.0.101", + "resolved": "https://registry.npmjs.org/@ai-sdk/google-vertex/-/google-vertex-5.0.101.tgz", + "integrity": "sha512-Lgi1Y2SuO1te3eWBsOqCoOvV+EDH1+K8r6OrNTMMexZd2Npyj6gPLODCn4bAwZeLA92lOm8zwCAqACOUEib+eQ==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/anthropic": "4.0.65", - "@ai-sdk/google": "4.0.82", - "@ai-sdk/openai-compatible": "3.0.57", - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49", + "@ai-sdk/anthropic": "4.0.71", + "@ai-sdk/google": "4.0.87", + "@ai-sdk/openai-compatible": "3.0.62", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53", "google-auth-library": "^10.6.2" }, "engines": { @@ -230,14 +230,14 @@ } }, "node_modules/@ai-sdk/openai": { - "version": "4.0.77", - "resolved": "https://registry.npmjs.org/@ai-sdk/openai/-/openai-4.0.77.tgz", - "integrity": "sha512-KUCeRneOIww8az/mH3dy68csmOTqa7s0F6fiHEkdXtr1u0xUoQTIU4y7k0YrjjBuyybpX2yN94rf9elhPu4zFg==", + "version": "4.0.83", + "resolved": "https://registry.npmjs.org/@ai-sdk/openai/-/openai-4.0.83.tgz", + "integrity": "sha512-NgqZVWoya7hfmtLkWyIjX9THbyYMLfxFqpShdAmvMnkoT+R9AZbdULGan8+4BiBWIZ1aHiAsc90l3SL8QU/0+Q==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -247,14 +247,14 @@ } }, "node_modules/@ai-sdk/openai-compatible": { - "version": "3.0.57", - "resolved": "https://registry.npmjs.org/@ai-sdk/openai-compatible/-/openai-compatible-3.0.57.tgz", - "integrity": "sha512-TEY5Ng/bTJXrYy9wpdxK70gABdYejYedaKbmPCZLhBkPe21rmLzkGxAGkfPYgAfCKPaoAO9THrEhmwUZ4Xhg2Q==", + "version": "3.0.62", + "resolved": "https://registry.npmjs.org/@ai-sdk/openai-compatible/-/openai-compatible-3.0.62.tgz", + "integrity": "sha512-IAz1xv8zot4KLGa8kN7UfAA6H5+ARzAgAKLsoKxHPBzZuOc2E3ohaELg0x/JR9HVTbMbiftb0dPyCb8cT2KBRg==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -264,9 +264,9 @@ } }, "node_modules/@ai-sdk/provider": { - "version": "4.0.18", - "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.18.tgz", - "integrity": "sha512-+GZJIgz1jk86pwEbb3f1BD2bdoSKyWE4Jg4YUc7NMnMozbWemSKYZcw2F4nMaO5qwsL5A8RmioAKW81YktRpIQ==", + "version": "4.0.21", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.21.tgz", + "integrity": "sha512-UpbC9C1oht8dhfKPbVXSLRZS3DI8uk8n5v2uMjBPNIYGsC2kL045ywq7D8Hh9KgnH9rP5W/GxQdYtWScRouqlA==", "license": "Apache-2.0", "dependencies": { "json-schema": "^0.4.0" @@ -276,12 +276,12 @@ } }, "node_modules/@ai-sdk/provider-utils": { - "version": "5.0.49", - "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.49.tgz", - "integrity": "sha512-T+/H8DCvqJoCqLhltVatS7ffA779Iyk/ZZrgNXkrgv51E0n2Cd6eIdCmtwhCGIKqtbTtwCvTmQdAMv5Z8rOGHQ==", + "version": "5.0.53", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.53.tgz", + "integrity": "sha512-VVe6UDd0y0/B4TfauuKKzcppqvhTmSluwwUVT4WG2/8h7gFbyFm7PM4UGM+ZXe3/RI6bv+Vl1cw/7aiBoQqdxg==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", + "@ai-sdk/provider": "4.0.21", "@standard-schema/spec": "^1.1.0", "@workflow/serde": "4.1.0", "eventsource-parser": "^3.0.8", @@ -295,14 +295,14 @@ } }, "node_modules/@ai-sdk/xai": { - "version": "5.0.10", - "resolved": "https://registry.npmjs.org/@ai-sdk/xai/-/xai-5.0.10.tgz", - "integrity": "sha512-fSZtjtoQPa6VR6a30WNGagbfIECWTJA1I5dJCovYOC/JDZqibpBNCkvjnxd2YaWJy0Ozzq52rWCS+D1sY3M/uQ==", + "version": "5.0.14", + "resolved": "https://registry.npmjs.org/@ai-sdk/xai/-/xai-5.0.14.tgz", + "integrity": "sha512-Omgf/lFOf1nHhRGlRQqD/t7DPllZaxen1kIGiMrgYfnhtfdM7YcKMUY76rw9CeFrOldGLfewbfShAm9hWAp5FA==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -805,9 +805,9 @@ } }, "node_modules/@modelcontextprotocol/sdk": { - "version": "1.30.1", - "resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.30.1.tgz", - "integrity": "sha512-H2HxLvC3HDNybePJaLdSrU1hhUK5iQw+WvV1b01myFyI7sdVGe1u/IPTE5D9fGCiJDVtgMV/lmFkQXLmQyIFYA==", + "version": "1.32.0", + "resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.32.0.tgz", + "integrity": "sha512-8BviX/hK4Gd2eL1KdTwm0gfl4d3wdoErW0Jd00h6nomUAtk6IJk5vwQJc6kfWtJTzN5OEV/lgIiJKI8fiDGAlA==", "license": "MIT", "dependencies": { "@hono/node-server": "^1.19.9 || ^2.0.5", @@ -1333,14 +1333,14 @@ } }, "node_modules/ai": { - "version": "7.0.116", - "resolved": "https://registry.npmjs.org/ai/-/ai-7.0.116.tgz", - "integrity": "sha512-gmkVGPzTNPcJixBiG9zUvP3gmvwko85dHiKrxFKK0zcVJIIu1FP6YDo3jna+ZrC2twMAsUPLr5htCwZjCINH3Q==", + "version": "7.0.127", + "resolved": "https://registry.npmjs.org/ai/-/ai-7.0.127.tgz", + "integrity": "sha512-JNsNPk4ZvRGEJuqdVo2lYFkm7uAuf3p/RwKFWn1Wh/+IK3U1RyFWw2GXoFnhzqNfxEuj7XJB4WpIUhl1sHToWg==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/gateway": "4.0.94", - "@ai-sdk/provider": "4.0.18", - "@ai-sdk/provider-utils": "5.0.49" + "@ai-sdk/gateway": "4.0.103", + "@ai-sdk/provider": "4.0.21", + "@ai-sdk/provider-utils": "5.0.53" }, "engines": { "node": ">=22" @@ -3225,9 +3225,9 @@ "license": "MIT" }, "node_modules/undici": { - "version": "7.29.0", - "resolved": "https://registry.npmjs.org/undici/-/undici-7.29.0.tgz", - "integrity": "sha512-IDxfleLmmbSskfWSUATiN1nfn2rDuvnMOqb5CWR92iIfojA0Ud+ulOAAEQ57LPr9rWmsreUyf5lwyao+7GNNVw==", + "version": "7.30.0", + "resolved": "https://registry.npmjs.org/undici/-/undici-7.30.0.tgz", + "integrity": "sha512-dkrQXeHSaoamnItlYbmzG0wFYrM0ZwDxCIg0A7aKjTyyhh9svRzCNFEzV+Vm05/yehjCzjDZ31KXfGEjYSztDQ==", "license": "MIT", "engines": { "node": ">=20.18.1" diff --git a/package.json b/package.json index c1b7a84..4d79c66 100644 --- a/package.json +++ b/package.json @@ -89,9 +89,9 @@ "prepublishOnly": "npm run typecheck && npm test && npm run build" }, "dependencies": { - "@modelcontextprotocol/sdk": "^1.30.1", + "@modelcontextprotocol/sdk": "^1.32.0", "@opentelemetry/api": "^1.9.1", - "ai": "^7.0.116", + "ai": "^7.0.127", "zod": "^4.6.5" }, "peerDependencies": { @@ -139,15 +139,15 @@ } }, "devDependencies": { - "@ai-sdk/amazon-bedrock": "^5.0.96", - "@ai-sdk/anthropic": "^4.0.65", - "@ai-sdk/azure": "^4.0.81", - "@ai-sdk/deepseek": "^3.0.54", - "@ai-sdk/google": "^4.0.82", - "@ai-sdk/google-vertex": "^5.0.95", - "@ai-sdk/openai": "^4.0.77", - "@ai-sdk/openai-compatible": "^3.0.57", - "@ai-sdk/xai": "^5.0.10", + "@ai-sdk/amazon-bedrock": "^5.0.105", + "@ai-sdk/anthropic": "^4.0.71", + "@ai-sdk/azure": "^4.0.90", + "@ai-sdk/deepseek": "^3.0.58", + "@ai-sdk/google": "^4.0.87", + "@ai-sdk/google-vertex": "^5.0.101", + "@ai-sdk/openai": "^4.0.83", + "@ai-sdk/openai-compatible": "^3.0.62", + "@ai-sdk/xai": "^5.0.14", "@types/node": "^22.9.0", "prettier": "^3.9.9", "tsup": "^8.5.1", diff --git a/src/agent.ts b/src/agent.ts new file mode 100644 index 0000000..0290ea2 --- /dev/null +++ b/src/agent.ts @@ -0,0 +1,619 @@ +import type { LanguageModel, ToolSet } from 'ai' +import { createApprovalController, isReadOnlyTool } from './approval.ts' +import { compactionMaxTokens } from './call-options.ts' +import { compactHistory, resolveCompaction } from './compaction.ts' +import type { IAgentInternalContext } from './internal.ts' +import { connectMcpServers, filterTools } from './mcp.ts' +import { buildModelFromStage, resolveStage } from './provider.ts' +import { runAgentLoop } from './runner.ts' +import { createSkillTools, validateSkills } from './skills.ts' +import { createFindToolsTool, FIND_TOOLS_TOOL } from './tool-search.ts' +import { wrapToolSet, type IToolWrapDeps } from './tool-wrap.ts' +import type { + AgentEvent, + EventHandler, + IAgentConfig, + IAgentRunOptions, + IAgentRunResult, + IConversationTurn, + IToolCatalogEntry, + LogLevel, + ToolApprovalMode, +} from './types.ts' + +export interface ICloseOptions { + // When true, close() polls until activeRuns reaches 0 (or timeoutMs elapses) + // before tearing down MCP connections. Default false: close immediately and + // let active runs fail mid-flight (the legacy behavior). + waitForRuns?: boolean + // Cap, in ms, applied to the entire close() call: + // 1. when waitForRuns is true: max time spent waiting for active runs to + // drain; + // 2. ALWAYS: max time spent on the underlying MCP transport teardown + // (some HTTP/SSE transports can hang on close if the peer is + // unresponsive). After this elapses close() resolves anyway and the + // transport is abandoned to GC. Default 30s. + timeoutMs?: number +} + +export interface ICompactOptions { + history: IConversationTurn[] + signal?: AbortSignal + // Compact even below the threshold. + force?: boolean +} + +export interface ICompactResult { + // The compacted history (the input array itself when nothing changed). + history: IConversationTurn[] + summary?: string + beforeTokens: number + afterTokens: number + compacted: boolean +} + +export interface IAgent { + // Multiple concurrent runs are supported on a single agent instance: each + // call gets its own runId via AsyncLocalStorage, its own usage accumulator, + // its own abort signal, and its own onEvent. Tools and models are shared. + // Reconnect throws if runs are in flight; close optionally waits. + run: (options: IAgentRunOptions) => Promise + // The filtered tool catalogue (MCP + native; built-ins excluded). + listTools: () => IToolCatalogEntry[] + // Configured skills (name + description). + listSkills: () => { name: string; description: string }[] + // Summarise all but the most recent turns of a history into one summary + // turn (the synthesizer model writes it). Never throws; the input array is + // never mutated. Without `force`, only above the compaction threshold. + compact: (options: ICompactOptions) => Promise + // The "autopilot toggle": applies to the next tool call, including calls + // of runs already in flight. + setToolApprovalMode: (mode: ToolApprovalMode) => void + getToolApprovalMode: () => ToolApprovalMode + // Drops the current MCP connections and reconnects with fresh headers + // (via getHeaders if configured). Throws if any runs are in progress. + reconnect: () => Promise + close: (options?: ICloseOptions) => Promise + activeRuns: () => number +} + +const LEVEL_RANK: Record = { + none: 0, + error: 1, + warn: 2, + info: 3, + debug: 4, +} + +export const createAgent = async ( + config: IAgentConfig, + baseEventHandler?: EventHandler, +): Promise => { + const debugMode = LEVEL_RANK[config.logLevel] >= LEVEL_RANK.debug + + const emit: EventHandler = (event: AgentEvent) => { + if (baseEventHandler) { + try { + baseEventHandler(event) + } catch (err) { + if (debugMode) { + // Surface handler errors only in debug mode; in normal mode they + // are silently swallowed to keep the agent robust to bad consumers. + // Write directly to console to avoid recursion via emit(). + console.error('[agent] event handler threw:', err) + } + } + } + } + + const log = (level: 'info' | 'warn' | 'error', message: string) => { + if (LEVEL_RANK[config.logLevel] >= LEVEL_RANK[level]) { + emit({ type: 'log', level, message }) + } + } + + // ── skills, approval, built-in tools ──────────────────────────────────── + // Validated up front: a bad skill is a configuration error, not something + // to discover mid-run. + const skills = validateSkills(config.skills) + // The request shows the input AFTER inputSanitizer (idempotent by + // contract); a throwing sanitizer redacts rather than leaks. + const sanitizeForApproval = async (name: string, input: unknown): Promise => { + if (!config.inputSanitizer) { + return input + } + try { + return await config.inputSanitizer(name, input) + } catch { + return '[input redacted: sanitizer failed]' + } + } + const approval = createApprovalController(config.toolApproval, { + sanitize: sanitizeForApproval, + fallbackEmit: emit, + }) + const compaction = resolveCompaction(config.compaction) + const wrapDeps: IToolWrapDeps = { + approval, + maxToolCalls: config.maxToolCalls, + maxToolOutputChars: compaction.maxToolOutputChars, + } + const builtinTools = wrapToolSet(createSkillTools(skills), wrapDeps, { builtIn: true }) + // find_tools exists whenever the strategy can resolve to 'search'. + const strategy = config.toolSelectionStrategy ?? 'auto' + const searchPossible = strategy === 'search' || strategy === 'auto' + let searchDisabled = false + // Live catalogue reader for find_tools (ctx is assigned below). + let catalogRef: () => IToolCatalogEntry[] = () => [] + const findTools = searchPossible + ? wrapToolSet( + createFindToolsTool(() => catalogRef()), + wrapDeps, + { builtIn: true }, + ) + : undefined + + // Built-in names are reserved: a host / MCP tool of the same name would + // be shadowed silently. 'auto' degrades instead of failing (it only means + // "search when the catalogue grows"), so existing configs keep working. + const assertNoBuiltinCollision = (tools: ToolSet): void => { + for (const name of Object.keys(builtinTools)) { + if (tools[name]) { + throw new Error(`Tool "${name}" collides with a built-in skill tool of the same name`) + } + } + if (findTools && tools[FIND_TOOLS_TOOL]) { + if (strategy === 'search') { + throw new Error( + `Tool "${FIND_TOOLS_TOOL}" collides with the built-in tool-search tool of the same name`, + ) + } + if (!searchDisabled) { + searchDisabled = true + log( + 'warn', + `[tools] a tool named "${FIND_TOOLS_TOOL}" shadows the built-in tool search; 'auto' stays on 'all'`, + ) + } + } + } + + // Forward declaration: connectMcpServers needs the onToolsChanged callback, + // and the callback must enqueue refreshes that drain only when activeRuns + // reaches zero. We set the impl after the run/close machinery is wired up. + let onToolsChanged: ((server: string) => void) | undefined + // Merge native tools (config.tools) into a freshly filtered MCP view. + // Called both at startup and after a tools/list_changed refresh, so native + // entries survive MCP catalog rebuilds. + const mergeNativeTools = ( + f: { + tools: ReturnType['tools'] + catalog: ReturnType['catalog'] + }, + onCollision: (name: string) => never, + ): { + tools: ReturnType['tools'] + catalog: ReturnType['catalog'] + } => { + if (!config.tools) { + return f + } + const nativeCatalog: ReturnType['catalog'] = [] + for (const [name, tool] of Object.entries(config.tools)) { + if (f.tools[name]) { + onCollision(name) + } + // availableTools wins over native registration, mirroring the MCP + // path: an explicit allowlist excludes everything not on it. + if (config.availableTools?.length && !config.availableTools.includes(name)) { + continue + } + if ( + !config.availableTools?.length && + config.excludedTools?.length && + config.excludedTools.includes(name) + ) { + continue + } + f.tools[name] = tool + const rawDesc = + typeof tool === 'object' && tool && 'description' in tool + ? (tool as { description?: unknown }).description + : '' + nativeCatalog.push({ + name, + description: typeof rawDesc === 'string' ? rawDesc : '', + server: '', + readOnly: isReadOnlyTool(tool), + }) + } + return { tools: f.tools, catalog: [...f.catalog, ...nativeCatalog] } + } + + const connect = async () => { + const c = await connectMcpServers( + config.mcpServers, + log, + config.clientName, + config.outputSanitizer, + (server) => onToolsChanged?.(server), + config.inputSanitizer, + { connectTimeoutMs: config.mcpConnectTimeoutMs }, + ) + const f = filterTools(c.tools, c.catalog, config.availableTools, config.excludedTools, log) + let merged: ReturnType + try { + merged = mergeNativeTools(f, (name) => { + throw new Error( + `Native tool "${name}" collides with an MCP-registered tool of the same name`, + ) + }) + assertNoBuiltinCollision(merged.tools) + } catch (err) { + // Tear the connection down so we don't leak open MCP transports when + // the caller's misconfiguration crashes startup. + void c.close().catch(() => {}) + throw err + } + return { + connection: c, + tools: wrapToolSet(merged.tools, wrapDeps), + catalog: merged.catalog, + } + } + + let connected: Awaited> + try { + connected = await connect() + } catch (err) { + emit({ + type: 'error', + error: err instanceof Error ? err : new Error(String(err)), + phase: 'init', + }) + throw err + } + + // Hard fail when explicitly requested and every configured server failed + // to connect. Without this, the agent would start with zero tools, the + // planner would produce a no-tool plan, and the user only sees the + // problem several seconds later when execution times out. We only + // enforce this when servers were configured at all - an agent that runs + // tool-less by design (mcpServers: {}) is a valid use case. + const configuredServers = Object.keys(config.mcpServers).length + if (config.failOnNoTools && configuredServers > 0) { + const anyConnected = connected.connection.results.some((r) => r.connected) + if (!anyConnected) { + const reasons = connected.connection.results + .filter((r) => !r.connected) + .map((r) => `${r.name}: ${r.error ?? 'unknown'}`) + .join('; ') + await connected.connection.close().catch(() => {}) + const err = new Error( + `All ${configuredServers} configured MCP server(s) failed to connect [${reasons}]`, + ) + emit({ type: 'error', error: err, phase: 'init' }) + throw err + } + } + + // Executor always inherits the top-level config; planner / synthesizer can + // override any field (provider, baseURL, apiKey, model) via the dedicated + // override blocks. Legacy plannerModel / synthesizerModel still work as + // model-only shortcuts when the override block is absent. + // + // Provider SDKs are dynamically imported (peerDependencies, optional), so + // model construction is async; build all three in parallel since they are + // independent. On failure (e.g. missing peer dep) we MUST close the MCP + // connection opened above - otherwise we leak open transports for what is + // typically a misconfiguration retry-loop. + let executorModel: LanguageModel + let plannerModel: LanguageModel + let synthesizerModel: LanguageModel + try { + const executorStage = resolveStage(config, undefined, undefined, 'executor') + const plannerStage = resolveStage(config, config.planner, config.plannerModel, 'planner') + const synthesizerStage = resolveStage( + config, + config.synthesizer, + config.synthesizerModel, + 'synthesizer', + ) + ;[executorModel, plannerModel, synthesizerModel] = await Promise.all([ + buildModelFromStage(config.clientName, executorStage), + buildModelFromStage(config.clientName, plannerStage), + buildModelFromStage(config.clientName, synthesizerStage), + ]) + } catch (err) { + await connected.connection.close().catch(() => {}) + emit({ + type: 'error', + error: err instanceof Error ? err : new Error(String(err)), + phase: 'init', + }) + throw err + } + + // Sorted by name, so the catalogue rendered into the (cached) planner prompt + // is the same whatever order the servers answered in. + const toEntries = (catalog: typeof connected.catalog): IToolCatalogEntry[] => + catalog + .map((c) => ({ + name: c.name, + description: c.description, + server: c.server, + readOnly: c.readOnly === true, + })) + .sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)) + + const ctx: IAgentInternalContext = { + config, + executorModel, + plannerModel, + synthesizerModel, + tools: connected.tools, + toolCatalog: toEntries(connected.catalog), + emit, + ...(Object.keys(builtinTools).length ? { builtinTools } : {}), + ...(skills.length ? { skills } : {}), + approval, + } + // Read through ctx so find_tools always searches the live catalogue + // (refresh / reconnect replace ctx.toolCatalog). + catalogRef = () => ctx.toolCatalog + const syncFindTools = (): void => { + if (findTools && !searchDisabled) { + ctx.findTools = findTools + } else { + delete ctx.findTools + } + } + syncFindTools() + + let activeRuns = 0 + let closed = false + let reconnecting = false + let refreshing = false + // Server names whose tools/list_changed has fired but whose refresh is + // deferred because activeRuns > 0 or another lifecycle op is in flight. + const pendingRefreshes = new Set() + + const applyRefresh = async (server: string): Promise => { + await connected.connection.refreshServer(server) + // Re-apply the availableTools/excludedTools filter to the freshly mutated + // raw maps, then mutate ctx.tools in place so existing closures pick up + // the new set. The synchronous delete+assign block runs without awaits, + // so no run() can interleave once activeRuns has been gated to 0. + const f = filterTools( + connected.connection.tools, + connected.connection.catalog, + config.availableTools, + config.excludedTools, + ) + // Native tools must be merged back in: filterTools/refreshServer only + // produce the MCP view, so without this they'd vanish from ctx.tools + // until the next reconnect. + const merged = mergeNativeTools(f, (name) => { + throw new Error(`Native tool "${name}" collides with an MCP-registered tool of the same name`) + }) + assertNoBuiltinCollision(merged.tools) + const wrapped = wrapToolSet(merged.tools, wrapDeps) + for (const k of Object.keys(ctx.tools)) { + delete ctx.tools[k] + } + Object.assign(ctx.tools, wrapped) + ctx.toolCatalog = toEntries(merged.catalog) + syncFindTools() + log( + 'info', + `[mcp] ${server}: tool list refreshed (${merged.catalog.length} tools total after filter)`, + ) + } + + const drainPendingRefreshes = async (): Promise => { + if (refreshing || closed || reconnecting) { + return + } + if (activeRuns > 0 || pendingRefreshes.size === 0) { + return + } + refreshing = true + try { + while (pendingRefreshes.size > 0 && activeRuns === 0 && !closed && !reconnecting) { + const next = pendingRefreshes.values().next().value as string + pendingRefreshes.delete(next) + try { + await applyRefresh(next) + } catch (err) { + log('warn', `[mcp] ${next}: refresh failed - ${(err as Error).message}`) + } + } + } finally { + refreshing = false + } + } + + // Wired up here so the closure captures the lifecycle flags and + // pendingRefreshes set declared above. + onToolsChanged = (server: string) => { + pendingRefreshes.add(server) + // Sync entry into drainPendingRefreshes runs to the first await before + // returning, so the refreshing/activeRuns gating is observed atomically + // from any run() call that lands after this notification. + void drainPendingRefreshes() + } + + return { + run: async (options: IAgentRunOptions) => { + if (closed) { + throw new Error('Agent is closed') + } + // Block new runs from racing into a mid-flight reconnect: the await on + // connect() inside reconnect() opens a window during which ctx.tools is + // about to be mutated. Accepting new runs in that window would let them + // observe a half-deleted ToolSet. + if (reconnecting) { + throw new Error('Agent is reconnecting; retry shortly') + } + if (refreshing) { + throw new Error('Agent is refreshing tools; retry shortly') + } + const cap = config.maxConcurrentRuns + if (typeof cap === 'number' && cap > 0 && activeRuns >= cap) { + // Synchronous reject: the caller should rate-limit on its side; we + // intentionally avoid a queue so close()/reconnect() stay simple + // (no pending-promises bookkeeping to drain). + throw new Error( + `maxConcurrentRuns reached (${activeRuns}/${cap}); rate-limit on the caller side or raise the cap`, + ) + } + activeRuns++ + try { + return await runAgentLoop(ctx, options) + } finally { + activeRuns-- + // Drain deferred tool refreshes once we go quiescent. Fire-and-forget: + // the next caller of run() either sees the refresh applied or gets + // the 'refreshing' rejection and retries. + void drainPendingRefreshes() + } + }, + listTools: () => ctx.toolCatalog.map((t) => ({ ...t })), + listSkills: () => skills.map((s) => ({ name: s.name, description: s.description })), + compact: async (options: ICompactOptions): Promise => { + const r = await compactHistory(options.history, ctx.synthesizerModel, { + thresholdTokens: compaction.thresholdTokens, + keepRecentTurns: compaction.keepRecentTurns, + summaryMaxTokens: compactionMaxTokens(config, compaction.summaryMaxTokens), + force: options.force, + signal: options.signal, + timeoutMs: config.llmTimeoutMs, + onUsage: (usage) => emit({ type: 'usage', phase: 'compact', usage }), + }) + if (r.compacted) { + emit({ + type: 'context.compacted', + scope: 'history', + beforeTokens: r.beforeTokens, + afterTokens: r.afterTokens, + }) + } + return { + history: r.history, + ...(r.summary ? { summary: r.summary } : {}), + beforeTokens: r.beforeTokens, + afterTokens: r.afterTokens, + compacted: r.compacted, + } + }, + setToolApprovalMode: (mode: ToolApprovalMode) => approval.setMode(mode), + getToolApprovalMode: () => approval.getMode(), + reconnect: async () => { + if (closed) { + throw new Error('Agent is closed') + } + if (reconnecting) { + throw new Error('Already reconnecting') + } + if (refreshing) { + throw new Error('Agent is refreshing tools; retry shortly') + } + if (activeRuns > 0) { + throw new Error(`Cannot reconnect with ${activeRuns} active run(s)`) + } + reconnecting = true + try { + const oldConnection = connected.connection + // Drop any deferred per-server refreshes - the new connection comes + // up with a fresh tool listing and its own subscriptions, so the + // pending entries are stale and would only re-trigger work. + pendingRefreshes.clear() + const fresh = await connect() + // Mutate ctx.tools in place so existing closures pick up new tools. + // The reconnecting flag held above blocks new run() calls during the + // await + mutation window, so no concurrent reader sees inconsistent + // state. + for (const k of Object.keys(ctx.tools)) { + delete ctx.tools[k] + } + Object.assign(ctx.tools, fresh.tools) + ctx.toolCatalog = toEntries(fresh.catalog) + syncFindTools() + connected = fresh + // Close old connections only after the new ones are wired up so a + // failed reconnect doesn't leave the agent without tools. + await oldConnection.close().catch(() => {}) + } finally { + reconnecting = false + } + }, + close: async (options?: ICloseOptions) => { + closed = true + const waitForRuns = options?.waitForRuns ?? false + const timeoutMs = options?.timeoutMs ?? 30_000 + // Single budget shared by both phases: waiting for runs to drain (if + // requested) and the MCP transport teardown. Whatever waitForRuns + // consumes is subtracted from what's available to the close itself, + // with a small floor so the transport always gets *some* chance. + const deadline = Date.now() + timeoutMs + if (waitForRuns && activeRuns > 0) { + // Poll instead of using EventEmitter to avoid coupling close() to + // event-handler ordering; activeRuns flips back to 0 after the run's + // try/finally regardless. + while (activeRuns > 0 && Date.now() < deadline) { + await new Promise((r) => setTimeout(r, 50)) + } + } + if (activeRuns > 0) { + emit({ + type: 'log', + level: 'warn', + message: `[agent] close called with ${activeRuns} active run(s); they will fail mid-flight`, + }) + } + // Race the MCP teardown against the remaining budget. An unresponsive + // HTTP/SSE peer can leave client.close() pending forever; the timeout + // guarantees close() itself resolves so callers (CLIs, test fixtures) + // never hang. Transports abandoned this way are GC'd when the agent is. + // + // Two tripwires we MUST get right: + // 1. clearTimeout when MCP wins the race - otherwise the timer keeps + // the event loop alive for the FULL closeBudget after close() + // returned, and processes that called close() at the end of main + // visibly hang. Verified manually: with 5s budget the process + // exited 5s late before this clear. + // 2. timeoutId.unref() - belt and suspenders so even an exception in + // the race body cannot pin the loop. + // + // No artificial floor: timeoutMs is a hard cap on the whole close() + // call (per ICloseOptions docs). If waitForRuns already burned the + // budget, we hand the teardown a 0ms slot - setTimeout(_, 0) still + // schedules to the next tick, so transports that finish synchronously + // can still win, but nothing here will exceed timeoutMs. + const closeBudget = Math.max(deadline - Date.now(), 0) + let timedOut = false + let timeoutId: ReturnType | undefined + await Promise.race([ + connected.connection.close().catch(() => {}), + new Promise((resolve) => { + timeoutId = setTimeout(() => { + timedOut = true + resolve() + }, closeBudget) + timeoutId.unref?.() + }), + ]) + if (timeoutId) { + clearTimeout(timeoutId) + } + if (timedOut) { + emit({ + type: 'log', + level: 'warn', + message: `[agent] MCP teardown exceeded ${closeBudget}ms; abandoning transport(s)`, + }) + } + }, + activeRuns: () => activeRuns, + } +} diff --git a/src/approval.ts b/src/approval.ts new file mode 100644 index 0000000..63c582c --- /dev/null +++ b/src/approval.ts @@ -0,0 +1,331 @@ +import { randomUUID } from 'node:crypto' +import { runContext } from './context.ts' +import type { + EventHandler, + IToolApprovalConfig, + IToolApprovalRequest, + ToolApprovalDecision, + ToolApprovalMode, + ToolPermission, +} from './types.ts' + +const MODES: ReadonlySet = new Set([ + 'autopilot', + 'ask-writes', + 'ask-all', + 'read-only', +]) + +export const isToolApprovalMode = (v: unknown): v is ToolApprovalMode => + typeof v === 'string' && MODES.has(v) + +/** + * Thrown from inside a tool's execute when the call was not approved. The AI + * SDK records it as a failed tool call (`ok: false`), so the replanner sees + * it; `toJSON` keeps the message when a trace is serialised. + */ +export class ToolDeniedError extends Error { + readonly reason?: string + + constructor(reason?: string) { + super( + `Tool call denied by the user${reason ? `: ${reason}` : ''}. Do not retry it; continue without it, or report what is blocked.`, + ) + this.name = 'ToolDeniedError' + this.reason = reason + } + + toJSON(): string { + return this.message + } +} + +// Read-only metadata on a tool object. MCP tools get it from +// annotations.readOnlyHint; native tools opt in with markReadOnly(tool). +export const markReadOnly = (tool: T, readOnly = true): T => { + ;(tool as { readOnly?: boolean }).readOnly = readOnly + return tool +} + +export const isReadOnlyTool = (tool: unknown): boolean => + Boolean(tool) && typeof tool === 'object' && (tool as { readOnly?: unknown }).readOnly === true + +const globCache = new Map() +const globToRegExp = (glob: string): RegExp => { + let re = globCache.get(glob) + if (!re) { + const body = glob + .split('*') + .map((part) => part.replace(/[.+?^${}()|[\]\\]/g, '\\$&')) + .join('.*') + re = new RegExp(`^${body}$`) + globCache.set(glob, re) + } + return re +} + +/** + * The most specific rule matching `name`: an exact key wins, then the + * longest matching '*' glob (ties: the first declared). Pure. + */ +export const matchToolRule = ( + rules: Record | undefined, + name: string, +): { pattern: string; permission: ToolPermission } | undefined => { + if (!rules) { + return undefined + } + if (Object.hasOwn(rules, name) && !name.includes('*')) { + return { pattern: name, permission: rules[name] } + } + let best: { pattern: string; permission: ToolPermission } | undefined + for (const [pattern, permission] of Object.entries(rules)) { + if (!pattern.includes('*')) { + continue + } + if (globToRegExp(pattern).test(name) && (!best || pattern.length > best.pattern.length)) { + best = { pattern, permission } + } + } + return best +} + +export interface IToolDecisionInput { + name: string + readOnly: boolean + mode: ToolApprovalMode + rules?: Record + remembered?: ReadonlySet + // Built-in tools (load_skill, read_skill_file, find_tools) are always allowed. + builtIn?: boolean +} + +export interface IToolDecision { + permission: ToolPermission + source: 'builtin' | 'remembered' | 'rule' | 'mode' + // The matched rule pattern (source 'rule'). + rule?: string +} + +/** + * Decide one call. Pure. Order: built-in -> remembered "always allow" -> + * most specific rule -> mode default (autopilot: allow; read-only: allow if + * readOnly else deny; ask-writes: allow if readOnly else ask; ask-all: ask). + */ +export const decideToolPermission = (input: IToolDecisionInput): IToolDecision => { + if (input.builtIn) { + return { permission: 'allow', source: 'builtin' } + } + if (input.remembered?.has(input.name)) { + return { permission: 'allow', source: 'remembered' } + } + const rule = matchToolRule(input.rules, input.name) + if (rule) { + return { permission: rule.permission, source: 'rule', rule: rule.pattern } + } + switch (input.mode) { + case 'autopilot': + return { permission: 'allow', source: 'mode' } + case 'read-only': + return { permission: input.readOnly ? 'allow' : 'deny', source: 'mode' } + case 'ask-writes': + return { permission: input.readOnly ? 'allow' : 'ask', source: 'mode' } + case 'ask-all': + return { permission: 'ask', source: 'mode' } + } +} + +export interface IApprovalGateOptions { + readOnly: boolean + builtIn?: boolean + signal?: AbortSignal +} + +export interface IApprovalController { + getMode: () => ToolApprovalMode + setMode: (mode: ToolApprovalMode) => void + // Tools approved with remember: true (agent-instance lifetime). + remembered: Set + // Resolves when the call may run; throws ToolDeniedError otherwise. + gate: (name: string, input: unknown, opts: IApprovalGateOptions) => Promise +} + +const normalizeDecision = ( + d: ToolApprovalDecision | undefined, +): { approved: boolean; reason?: string; remember?: boolean } => + typeof d === 'boolean' + ? { approved: d } + : d && typeof d === 'object' + ? { approved: d.approved === true, reason: d.reason, remember: d.remember } + : { approved: false, reason: 'invalid approval decision' } + +class ApprovalTimeout extends Error {} + +const awaitDecision = ( + pending: Promise, + timeoutMs: number | undefined, + signal: AbortSignal | undefined, +): Promise => + new Promise((resolve, reject) => { + let timer: ReturnType | undefined + const done = (): void => { + if (timer) { + clearTimeout(timer) + } + signal?.removeEventListener('abort', onAbort) + } + function onAbort(): void { + done() + reject(signal?.reason ?? new DOMException('Aborted', 'AbortError')) + } + if (signal?.aborted) { + onAbort() + return + } + signal?.addEventListener('abort', onAbort, { once: true }) + if (typeof timeoutMs === 'number' && timeoutMs > 0) { + timer = setTimeout(() => { + done() + reject(new ApprovalTimeout()) + }, timeoutMs) + } + pending.then( + (v) => { + done() + resolve(v) + }, + (err: unknown) => { + done() + reject(err) + }, + ) + }) + +/** + * One controller per agent instance. The gate runs INSIDE each tool's + * execute, so it covers native and MCP tools alike; it reads the run's + * emit / current step / runId from the AsyncLocalStorage run context. + */ +export const createApprovalController = ( + config: IToolApprovalConfig | undefined, + deps: { + // Applied to the input shown in the request (idempotent inputSanitizer). + sanitize?: (name: string, input: unknown) => Promise + // Used when a tool is called outside of a run. + fallbackEmit?: EventHandler + } = {}, +): IApprovalController => { + let mode: ToolApprovalMode = config?.mode ?? 'autopilot' + if (!isToolApprovalMode(mode)) { + throw new Error( + `toolApproval.mode must be ${[...MODES].join(' | ')}, got: ${JSON.stringify(mode)}`, + ) + } + const remembered = new Set() + + const gate = async (name: string, input: unknown, opts: IApprovalGateOptions): Promise => { + const decision = decideToolPermission({ + name, + readOnly: opts.readOnly, + mode, + rules: config?.rules, + remembered, + builtIn: opts.builtIn, + }) + if (decision.permission === 'allow') { + return + } + const store = runContext.getStore() + const emit: EventHandler = store?.emit ?? deps.fallbackEmit ?? (() => {}) + const id = randomUUID() + const deny = (reason: string, automatic: boolean): never => { + emit({ type: 'tool.approval-resolved', id, name, approved: false, reason, automatic }) + throw new ToolDeniedError(reason) + } + if (decision.permission === 'deny') { + deny( + decision.source === 'rule' + ? `denied by rule "${decision.rule}"` + : `"${mode}" mode allows read-only tools only`, + true, + ) + } + const onRequest = config?.onRequest + if (!onRequest) { + deny('no approval handler configured', true) + return + } + const shown = deps.sanitize ? await deps.sanitize(name, input) : input + const step = store?.currentStep + emit({ + type: 'tool.approval-requested', + id, + name, + input: shown, + readOnly: opts.readOnly, + ...(step ? { step } : {}), + }) + const request: IToolApprovalRequest = { + id, + toolName: name, + input: shown, + readOnly: opts.readOnly, + ...(step ? { step } : {}), + ...(store?.runId ? { runId: store.runId } : {}), + } + let raw: ToolApprovalDecision + try { + raw = await awaitDecision( + Promise.resolve().then(() => onRequest(request)), + config?.timeoutMs, + opts.signal, + ) + } catch (err) { + if (err instanceof ApprovalTimeout) { + deny(`no decision within ${config?.timeoutMs}ms`, true) + } + if (opts.signal?.aborted) { + emit({ + type: 'tool.approval-resolved', + id, + name, + approved: false, + reason: 'aborted', + automatic: true, + }) + throw err + } + deny(`approval handler failed: ${(err as Error)?.message ?? String(err)}`, true) + return + } + const d = normalizeDecision(raw) + if (d.approved && d.remember) { + remembered.add(name) + } + emit({ + type: 'tool.approval-resolved', + id, + name, + approved: d.approved, + ...(d.reason ? { reason: d.reason } : {}), + automatic: false, + }) + if (!d.approved) { + throw new ToolDeniedError(d.reason) + } + } + + return { + getMode: () => mode, + setMode: (next) => { + if (!isToolApprovalMode(next)) { + throw new Error( + `tool approval mode must be ${[...MODES].join(' | ')}, got: ${JSON.stringify(next)}`, + ) + } + mode = next + }, + remembered, + gate, + } +} diff --git a/src/caching.ts b/src/caching.ts new file mode 100644 index 0000000..3c6bca9 --- /dev/null +++ b/src/caching.ts @@ -0,0 +1,108 @@ +import type { SystemModelMessage } from 'ai' +import type { PromptCachingSetting } from './types.ts' +import type { ProviderOptionsMap } from './thinking.ts' + +export interface IResolvedPromptCaching { + enabled: boolean + ttl?: '5m' | '1h' + key?: string +} + +// Default ON: a cache breakpoint on a run-stable system prompt is free on +// providers that ignore it and cuts input cost on the ones that honour it. +export const resolvePromptCaching = ( + setting: PromptCachingSetting | undefined, +): IResolvedPromptCaching => { + if (setting === false) { + return { enabled: false } + } + if (setting === undefined || setting === true) { + return { enabled: true } + } + return { + enabled: true, + ...(setting.ttl ? { ttl: setting.ttl } : {}), + ...(setting.key ? { key: setting.key } : {}), + } +} + +/** + * The `instructions` call option for a system prompt. With caching on, a + * SystemModelMessage carrying an Anthropic ephemeral cache breakpoint (other + * providers ignore the key); otherwise the plain string. + */ +export const buildInstructions = ( + system: string, + caching: IResolvedPromptCaching, +): string | SystemModelMessage => { + if (!caching.enabled) { + return system + } + return { + role: 'system', + content: system, + providerOptions: { + anthropic: { + cacheControl: { type: 'ephemeral', ...(caching.ttl ? { ttl: caching.ttl } : {}) }, + }, + }, + } +} + +// Call-level provider options for caching: an OpenAI prompt cache key per +// stage so requests of the same stage share a cache shard. +export const cacheProviderOptions = ( + caching: IResolvedPromptCaching, + clientName: string, + stage: string, +): ProviderOptionsMap | undefined => + caching.enabled + ? { openai: { promptCacheKey: caching.key ?? `${clientName}:${stage}` } } + : undefined + +type WithProviderOptions = { providerOptions?: unknown } + +// The message without an Anthropic cache breakpoint (other provider options kept). +const withoutBreakpoint = (m: M): M => { + const own = m.providerOptions as Record> | undefined + if (!own?.anthropic || !('cacheControl' in own.anthropic)) { + return m + } + const { cacheControl: _drop, ...anthropic } = own.anthropic + const { anthropic: _old, ...rest } = own + const providerOptions = Object.keys(anthropic).length ? { ...rest, anthropic } : rest + const { providerOptions: _po, ...base } = m + return (Object.keys(providerOptions).length ? { ...base, providerOptions } : base) as M +} + +/** + * The conversation with a rolling cache breakpoint on its LAST message — the + * agent-loop pattern Claude Code uses: every tool-calling round re-sends the + * rounds before it, and with the breakpoint on the newest message each request + * reads all of them from the cache and writes only the new tail. Breakpoints + * on earlier messages are removed (the SDK carries a round's messages into the + * next one), so a request holds at most two — system + newest; Anthropic + * rejects more than four. + */ +export const withRollingBreakpoint = ( + messages: M[], + caching: IResolvedPromptCaching, +): M[] => { + if (!caching.enabled || messages.length === 0) { + return messages + } + const out = messages.map((m, i) => (i < messages.length - 1 ? withoutBreakpoint(m) : m)) + const last = out[out.length - 1] + const own = (last.providerOptions ?? {}) as Record> + out[out.length - 1] = { + ...last, + providerOptions: { + ...own, + anthropic: { + ...own.anthropic, + cacheControl: { type: 'ephemeral', ...(caching.ttl ? { ttl: caching.ttl } : {}) }, + }, + }, + } + return out +} diff --git a/src/call-options.ts b/src/call-options.ts new file mode 100644 index 0000000..9816a86 --- /dev/null +++ b/src/call-options.ts @@ -0,0 +1,50 @@ +import type { JSONValue, SystemModelMessage } from 'ai' +import { buildInstructions, cacheProviderOptions, resolvePromptCaching } from './caching.ts' +import { resolveLimits } from './limits.ts' +import { mergeProviderOptions, resolveThinking, stageThinkingSetting } from './thinking.ts' +import type { AgentStage, IAgentConfig, ThinkingLevel } from './types.ts' + +type CallProviderOptions = Record> + +export interface IStageCallOptions { + instructions: string | SystemModelMessage + reasoning?: ThinkingLevel + providerOptions?: CallProviderOptions + maxOutputTokens?: number +} + +const positive = (v: number | undefined): number | undefined => + typeof v === 'number' && Number.isFinite(v) && v > 0 ? Math.floor(v) : undefined + +/** + * The per-call options every stage passes to the AI SDK: the system prompt + * as `instructions` (with a cache breakpoint when prompt caching is on), the + * stage's thinking (portable `reasoning` + provider options), the prompt + * cache key, and the stage's per-call output cap. Thinking provider options + * are merged first, caching ones on top (deep, per provider key). + */ +export const stageCallOptions = ( + config: IAgentConfig, + stage: AgentStage, + system: string, +): IStageCallOptions => { + const caching = resolvePromptCaching(config.promptCaching) + const thinking = resolveThinking(stageThinkingSetting(config, stage)) + const providerOptions = mergeProviderOptions( + thinking.providerOptions, + cacheProviderOptions(caching, config.clientName, stage), + ) as CallProviderOptions | undefined + const maxOutputTokens = positive(resolveLimits(config).perCall?.[stage]) + return { + instructions: buildInstructions(system, caching), + ...(thinking.reasoning ? { reasoning: thinking.reasoning } : {}), + ...(providerOptions ? { providerOptions } : {}), + ...(maxOutputTokens ? { maxOutputTokens } : {}), + } +} + +// The output cap of a compaction summary call. +export const compactionMaxTokens = (config: IAgentConfig, summaryMaxTokens: number): number => { + const perCall = positive(resolveLimits(config).perCall?.compaction) + return perCall ? Math.min(perCall, summaryMaxTokens) : summaryMaxTokens +} diff --git a/src/cli/args.ts b/src/cli/args.ts index 425d6b2..841a941 100644 --- a/src/cli/args.ts +++ b/src/cli/args.ts @@ -21,7 +21,7 @@ Options: --log-level= AGENT_LOG_LEVEL (none|error|warn|info|debug) --max-iterations= AGENT_MAX_ITERATIONS --max-steps-per-task= AGENT_MAX_STEPS_PER_TASK - --tool-strategy= AGENT_TOOL_SELECTION_STRATEGY (all|plan-narrowed) + --tool-strategy= AGENT_TOOL_SELECTION_STRATEGY (auto|all|plan-narrowed|search) -h, --help Show this help API keys are still read from env (AGENT_API_KEY etc.) - we deliberately do @@ -34,8 +34,26 @@ Required env vars (set directly or via --env-file): AGENT_MODEL model id MCP_SERVERS JSON: { "": { "url": "...", "headers"?: {...} } | { "command": "...", "args"?: [], "env"?: {} } } +Optional env vars (see env.example for the full list): + AGENT_THINKING off | minimal | low | medium | high | xhigh | + AGENT_MAX_INPUT_TOKENS per-run caps; crossing one jumps to the final answer + AGENT_MAX_OUTPUT_TOKENS + AGENT_MAX_REASONING_TOKENS + AGENT_MAX_TOTAL_TOKENS + AGENT_MAX_TOOL_CALLS cap on tool calls per run + AGENT_TOOL_APPROVAL autopilot | ask-writes | ask-all | read-only + AGENT_SKILLS_DIR folder of /SKILL.md skills + AGENT_CONTEXT_WINDOW_TOKENS model window (auto-compaction threshold = 50%) + AGENT_COMPACTION off disables automatic compaction + Commands inside the REPL: /status, /tools, /history, /reset, /reconnect, /exit + /compact summarise the conversation history now + /autopilot toggle autopilot (no tool approval prompts) + /approval autopilot | ask-writes | ask-all | read-only + +When the approval mode asks, answer y (allow once), n (deny) or a (always +allow this tool for the session). ` const FLAG_TO_ENV: Record = { diff --git a/src/cli/config.ts b/src/cli/config.ts index 0403d04..411c284 100644 --- a/src/cli/config.ts +++ b/src/cli/config.ts @@ -1,9 +1,14 @@ import type { IAgentConfig, IAgentStageOverride, + ICompactionConfig, IMcpServerConfig, + ITokenLimits, LogLevel, ProviderType, + ThinkingLevel, + ThinkingSetting, + ToolApprovalMode, } from '../index.ts' const PROVIDERS: readonly ProviderType[] = [ @@ -24,7 +29,22 @@ const REQUIRES_BASE_URL: ReadonlySet = new Set([ 'azure', ]) const LOG_LEVELS: readonly LogLevel[] = ['none', 'error', 'warn', 'info', 'debug'] -const TOOL_STRATEGIES = ['all', 'plan-narrowed'] as const +const TOOL_STRATEGIES = ['all', 'plan-narrowed', 'search', 'auto'] as const +const THINKING_LEVELS: readonly ThinkingLevel[] = [ + 'provider-default', + 'none', + 'minimal', + 'low', + 'medium', + 'high', + 'xhigh', +] +const APPROVAL_MODES: readonly ToolApprovalMode[] = [ + 'autopilot', + 'ask-writes', + 'ask-all', + 'read-only', +] type ToolStrategy = (typeof TOOL_STRATEGIES)[number] export const loadConfig = (): IAgentConfig => { @@ -66,6 +86,11 @@ export const loadConfig = (): IAgentConfig => { maxStepsPerTask: parsePositiveInt(process.env.AGENT_MAX_STEPS_PER_TASK, 8), maxRevisions: parseOptionalInt(process.env.AGENT_MAX_REVISIONS), maxTotalTokens: parseOptionalInt(process.env.AGENT_MAX_TOTAL_TOKENS), + limits: parseLimits(), + maxToolCalls: parseOptionalInt(process.env.AGENT_MAX_TOOL_CALLS), + thinking: parseThinking(process.env.AGENT_THINKING), + toolApproval: parseApproval(process.env.AGENT_TOOL_APPROVAL), + compaction: parseCompaction(), llmTimeoutMs: parseOptionalInt(process.env.AGENT_LLM_TIMEOUT_MS), llmMaxRetries: parseOptionalInt(process.env.AGENT_LLM_MAX_RETRIES), toolSelectionStrategy: parseToolStrategy(process.env.AGENT_TOOL_SELECTION_STRATEGY), @@ -73,6 +98,87 @@ export const loadConfig = (): IAgentConfig => { } } +// AGENT_THINKING: off | provider-default | none | minimal | low | medium | +// high | xhigh | (an exact budget in tokens, level 'medium'). +// 'off' disables thinking explicitly (reasoning: 'none'); unset leaves the +// provider default. +export const parseThinking = (raw: string | undefined): ThinkingSetting | undefined => { + const v = raw?.trim().toLowerCase() + if (!v) { + return undefined + } + if (v === 'off') { + return 'none' + } + if (THINKING_LEVELS.includes(v as ThinkingLevel)) { + return v as ThinkingLevel + } + const n = Number(v) + if (Number.isFinite(n) && n > 0) { + return { level: 'medium', budgetTokens: Math.floor(n) } + } + throw new Error( + `AGENT_THINKING must be off | ${THINKING_LEVELS.join(' | ')} | , got: ${raw}`, + ) +} + +// AGENT_MAX_INPUT_TOKENS / AGENT_MAX_OUTPUT_TOKENS / AGENT_MAX_REASONING_TOKENS. +// (AGENT_MAX_TOTAL_TOKENS keeps feeding the legacy top-level maxTotalTokens.) +const parseLimits = (): ITokenLimits | undefined => { + const limits: ITokenLimits = {} + const input = parseOptionalInt(process.env.AGENT_MAX_INPUT_TOKENS) + const output = parseOptionalInt(process.env.AGENT_MAX_OUTPUT_TOKENS) + const reasoning = parseOptionalInt(process.env.AGENT_MAX_REASONING_TOKENS) + if (input) { + limits.maxInputTokens = input + } + if (output) { + limits.maxOutputTokens = output + } + if (reasoning) { + limits.maxReasoningTokens = reasoning + } + return Object.keys(limits).length ? limits : undefined +} + +const parseApproval = (raw: string | undefined): { mode: ToolApprovalMode } | undefined => { + const v = raw?.trim() + if (!v) { + return undefined + } + if (!APPROVAL_MODES.includes(v as ToolApprovalMode)) { + throw new Error(`AGENT_TOOL_APPROVAL must be ${APPROVAL_MODES.join(' | ')}, got: ${v}`) + } + return { mode: v as ToolApprovalMode } +} + +// AGENT_COMPACTION=off disables automatic compaction (the /compact command +// still works); AGENT_CONTEXT_WINDOW_TOKENS sets the window the default +// threshold (50%) is derived from. +const parseCompaction = (): ICompactionConfig | undefined => { + const raw = process.env.AGENT_COMPACTION?.trim().toLowerCase() + const window = parseOptionalInt(process.env.AGENT_CONTEXT_WINDOW_TOKENS) + const cfg: ICompactionConfig = {} + if (raw) { + if (['off', 'false', '0', 'no'].includes(raw)) { + cfg.auto = false + } else if (['on', 'auto', 'true', '1', 'yes'].includes(raw)) { + cfg.auto = true + } else { + throw new Error(`AGENT_COMPACTION must be on | off, got: ${process.env.AGENT_COMPACTION}`) + } + } + if (window) { + cfg.contextWindowTokens = window + } + return Object.keys(cfg).length ? cfg : undefined +} + +// AGENT_SKILLS_DIR: a folder of /SKILL.md (loaded by the REPL, which +// is async; loadConfig stays sync). +export const skillsDirFromEnv = (): string | undefined => + process.env.AGENT_SKILLS_DIR?.trim() || undefined + // Read AGENT__PROVIDER_TYPE / _BASE_URL / _API_KEY / _MODEL into a // stage override block. Returns undefined if no field is set so the agent // inherits everything from the top-level defaults. diff --git a/src/cli/run.ts b/src/cli/run.ts index 01d7473..2f5768f 100644 --- a/src/cli/run.ts +++ b/src/cli/run.ts @@ -1,10 +1,28 @@ import { stdin as input, stdout as output } from 'node:process' import { createInterface } from 'node:readline/promises' -import type { AgentEvent, IConversationTurn } from '../index.ts' -import { createAgent } from '../index.ts' -import { loadConfig } from './config.ts' +import type { + AgentEvent, + IConversationTurn, + IToolApprovalRequest, + ThinkingSetting, + ToolApprovalDecision, + ToolApprovalMode, +} from '../index.ts' +import { createAgent, loadSkillsFromDir } from '../index.ts' +import { isSummaryTurn } from '../compaction.ts' +import { loadConfig, skillsDirFromEnv } from './config.ts' const HISTORY_LIMIT = 16 +const APPROVAL_MODES: readonly ToolApprovalMode[] = [ + 'autopilot', + 'ask-writes', + 'ask-all', + 'read-only', +] + +// ANSI dim for the model's thoughts; plain text when stdout is not a TTY. +const DIM = output.isTTY ? '\x1b[2m' : '' +const UNDIM = output.isTTY ? '\x1b[22m' : '' const truncate = (s: string, n = 240): string => { const flat = s.replace(/\s+/g, ' ').trim() @@ -19,8 +37,34 @@ const formatJson = (v: unknown): string => { } } +const describeThinking = (t: ThinkingSetting | undefined): string => { + if (t === undefined || t === false) { + return 'provider default' + } + if (t === true) { + return 'medium' + } + if (typeof t === 'string') { + return t + } + return `${t.level ?? 'medium'}${t.budgetTokens ? ` (budget ${t.budgetTokens})` : ''}` +} + +// Keep the REPL history bounded: drop the oldest turns but never the +// compaction summary at the head (it stands for everything before it). +const trimHistory = (history: IConversationTurn[]): void => { + while (history.length > HISTORY_LIMIT) { + const at = history.length && isSummaryTurn(history[0]) ? 1 : 0 + history.splice(at, 1) + } +} + export const runRepl = async (): Promise => { const config = loadConfig() + const skillsDir = skillsDirFromEnv() + if (skillsDir) { + config.skills = await loadSkillsFromDir(skillsDir) + } const mcpNames = Object.keys(config.mcpServers) // Resolve effective per-stage models: the override block wins over the @@ -33,21 +77,34 @@ export const runRepl = async (): Promise => { console.log( `[session] maxIterations=${config.maxIterations} maxStepsPerTask=${config.maxStepsPerTask} timeoutMs=${config.llmTimeoutMs ?? 'none'} retries=${config.llmMaxRetries ?? 2}`, ) + console.log( + `[session] thinking=${describeThinking(config.thinking)} approval=${config.toolApproval?.mode ?? 'autopilot'} skills=${config.skills?.length ?? 0}`, + ) console.log(`[mcp] servers=${mcpNames.length ? mcpNames.join(', ') : '(none)'}`) - console.log('[hint] commands: /status, /tools, /history, /reset, /reconnect, /exit\n') + console.log( + '[hint] commands: /status, /tools, /history, /reset, /compact, /autopilot, /approval , /reconnect, /exit\n', + ) - let streamingFinal = false - let streamingThought = false + // Which stream (if any) currently owns the output line. + let streaming: 'none' | 'thought' | 'reasoning' | 'final' = 'none' + const endLine = (): void => { + if (streaming !== 'none') { + process.stdout.write(streaming === 'reasoning' ? `${UNDIM}\n` : '\n') + streaming = 'none' + } + } const onEvent = (event: AgentEvent): void => { switch (event.type) { case 'log': + endLine() console.log(`[${event.level}] ${event.message}`) break case 'plan.thought-delta': - if (!streamingThought) { + if (streaming !== 'thought') { + endLine() process.stdout.write('\n[plan] ') - streamingThought = true + streaming = 'thought' } process.stdout.write(event.delta) break @@ -56,17 +113,14 @@ export const runRepl = async (): Promise => { // to display. The same step may be revised before plan.created lands; // we re-render from plan.created when it arrives (cleaner than tracking // partial->final diffs in the REPL). - if (streamingThought) { - process.stdout.write('\n') - streamingThought = false - } + endLine() console.log(` ${event.index + 1}. ${event.step.description}`) break case 'plan.created': - if (streamingThought) { - process.stdout.write('\n') - streamingThought = false + if (streaming === 'thought') { + endLine() } else { + endLine() console.log(`\n[plan] ${event.plan.thought}`) } for (const [i, s] of event.plan.steps.entries()) { @@ -75,29 +129,75 @@ export const runRepl = async (): Promise => { } break case 'plan.revised': + endLine() console.log(`\n[replan] reason=${event.reason}`) for (const [i, s] of event.plan.steps.entries()) { console.log(` ${i + 1}. ${s.description}`) } break + case 'skill.activated': + endLine() + console.log(`[skill] ${event.name} (by ${event.by})`) + break case 'step.start': + endLine() console.log(`\n[step ${event.index + 1}] ${event.step.description}`) break + case 'step.reasoning-delta': + case 'final.reasoning-delta': + if (streaming !== 'reasoning') { + endLine() + process.stdout.write(`${DIM}[think] `) + streaming = 'reasoning' + } + process.stdout.write(event.delta) + break case 'step.tool-call': + endLine() console.log(` -> ${event.name} ${formatJson(event.input)}`) break case 'step.tool-result': { + endLine() const status = event.ok ? 'ok' : 'fail' console.log(` <- ${event.name} ${status} ${formatJson(event.output)}`) break } + case 'tools.discovered': + endLine() + console.log( + ` ?? find_tools "${event.query}" -> ${event.names.length ? event.names.join(', ') : '(nothing)'}`, + ) + break + case 'tool.approval-resolved': + // Prompted decisions are visible at the prompt; show the automatic ones. + if (event.automatic && !event.approved) { + endLine() + console.log(` !! ${event.name} denied: ${event.reason ?? 'no reason'}`) + } + break + case 'subagent.start': + endLine() + console.log(` >> subagent ${event.name}: ${truncate(event.task, 160)}`) + break + case 'subagent.complete': + endLine() + console.log( + ` << subagent ${event.name} done (${event.usage.totalTokens} tokens): ${truncate(event.text, 160)}`, + ) + break + case 'subagent.error': + endLine() + console.log(` << subagent ${event.name} failed: ${event.error}`) + break case 'step.complete': + endLine() console.log( ` = ${truncate(event.result.summary, 400)} (${event.result.durationMs}ms, ${event.result.toolCalls.length} tool calls)`, ) break case 'replan.decision': if (event.mode !== 'continue') { + endLine() console.log(`[replan] ${event.mode} (${event.cause}): ${event.reason}`) } break @@ -108,30 +208,39 @@ export const runRepl = async (): Promise => { case 'retry': // If the planner is retrying, the partial thought we already streamed // is no longer authoritative - flush the line and reset the flag. - if (event.phase === 'plan' && streamingThought) { - process.stdout.write('\n') - streamingThought = false - } + endLine() console.log(`[retry] ${event.phase} attempt=${event.attempt}: ${event.error}`) break case 'budget.exceeded': - console.log(`[budget] token cap reached: ${event.tokens}/${event.cap} - finishing early`) + endLine() + console.log( + `[budget] ${event.kind} cap reached: ${event.tokens}/${event.cap} - finishing early`, + ) break case 'revisions.exceeded': + endLine() console.log(`[budget] max ${event.cap} replan-revisions reached - finishing early`) break + case 'context.compacted': + endLine() + console.log( + `[compact] ${event.scope}: ~${event.beforeTokens} -> ~${event.afterTokens} tokens`, + ) + break case 'final.text-delta': - if (!streamingFinal) { + if (streaming !== 'final') { + endLine() process.stdout.write('\nassistant> ') - streamingFinal = true + streaming = 'final' } process.stdout.write(event.delta) break case 'final': - if (streamingFinal) { + if (streaming === 'final') { process.stdout.write('\n\n') - streamingFinal = false + streaming = 'none' } else { + endLine() console.log(`\nassistant> ${event.text}\n`) } break @@ -139,32 +248,61 @@ export const runRepl = async (): Promise => { // A planner failure may interrupt mid-stream. Flush the partial // [plan] line so the next output (fallback plan or error) starts // cleanly, mirroring what the retry case does. - if (event.phase === 'plan' && streamingThought) { - process.stdout.write('\n') - streamingThought = false - } + endLine() console.error(`[error] (${event.phase}) ${event.error.message}`) break } } + const rl = createInterface({ input, output }) + let runController: AbortController | null = null + let inputController: AbortController | null = null + + // Approval prompts go through the REPL's readline. Parallel tool calls + // would otherwise interleave questions, so they are asked one at a time. + let approvalQueue: Promise = Promise.resolve() + const askApproval = (req: IToolApprovalRequest): Promise => { + const ask = async (): Promise => { + endLine() + const answer = ( + await rl.question( + `[approve] ${req.toolName} ${formatJson(req.input)}${req.readOnly ? ' (read-only)' : ''}\n allow? [y]es / [n]o / [a]lways: `, + runController ? { signal: runController.signal } : {}, + ) + ) + .trim() + .toLowerCase() + if (answer === 'a' || answer === 'always') { + return { approved: true, remember: true } + } + if (answer === 'y' || answer === 'yes') { + return true + } + return { approved: false, reason: 'denied at the prompt' } + } + const next = approvalQueue.then(ask, ask) + approvalQueue = next.catch(() => {}) + return next + } + config.toolApproval = { ...config.toolApproval, onRequest: askApproval } + const agent = await createAgent(config, onEvent) + // /autopilot toggles back to the last non-autopilot mode. + let lastAskMode: ToolApprovalMode = + agent.getToolApprovalMode() === 'autopilot' ? 'ask-writes' : agent.getToolApprovalMode() const tools = agent.listTools() console.log( `[tools] available=${tools.length}${tools.length ? `: ${tools.map((t) => t.name).join(', ')}` : ''}`, ) - - const rl = createInterface({ input, output }) const history: IConversationTurn[] = [] - let runController: AbortController | null = null - let inputController: AbortController | null = null // Use process-level SIGINT instead of rl.on('SIGINT'): the latter only fires // while rl.question() is actively reading. During `await agent.run(...)` // readline is idle, so Ctrl-C would otherwise hit Node's default handler // (terminate). With this listener: - // - Ctrl-C during a run: aborts the run via runController. + // - Ctrl-C during a run: aborts the run via runController (an open + // approval question is cancelled with it). // - Ctrl-C at the prompt: aborts the rl.question() via inputController, // which makes the await reject with AbortError - the loop catches it // and breaks cleanly. (Just calling rl.close() does NOT always reject @@ -191,6 +329,10 @@ export const runRepl = async (): Promise => { } } process.on('SIGINT', onSigInt) + // While readline is reading (the prompt, an approval question) the TTY is + // in raw mode and Ctrl-C reaches readline instead of the process; route it + // to the same handler (only one of the two ever fires for a keypress). + rl.on('SIGINT', onSigInt) try { while (true) { @@ -220,17 +362,40 @@ export const runRepl = async (): Promise => { console.log('[tools] (none)') } else { for (const t of tools) { - console.log(` - ${t.name}: ${truncate(t.description, 200)}`) + console.log( + ` - ${t.name}${t.readOnly ? ' (read-only)' : ''}: ${truncate(t.description, 200)}`, + ) } } continue } if (prompt === '/status') { + const limits = config.limits ?? {} + const caps = [ + limits.maxInputTokens ? `in=${limits.maxInputTokens}` : '', + limits.maxOutputTokens ? `out=${limits.maxOutputTokens}` : '', + limits.maxReasoningTokens ? `reasoning=${limits.maxReasoningTokens}` : '', + (limits.maxTotalTokens ?? config.maxTotalTokens) + ? `total=${limits.maxTotalTokens ?? config.maxTotalTokens}` + : '', + config.maxToolCalls ? `toolCalls=${config.maxToolCalls}` : '', + ].filter(Boolean) console.log( `[status] model=${config.model} planner=${plannerModelEff} synth=${synthModelEff}`, ) console.log( - `[status] tools=${tools.length} mcp=${mcpNames.join(', ') || '(none)'} historyTurns=${history.length}`, + `[status] tools=${tools.length} strategy=${config.toolSelectionStrategy ?? 'auto'} mcp=${mcpNames.join(', ') || '(none)'} historyTurns=${history.length}`, + ) + console.log( + `[status] thinking=${describeThinking(config.thinking)} approval=${agent.getToolApprovalMode()} limits=${caps.length ? caps.join(' ') : '(none)'}`, + ) + console.log( + `[status] compaction=${config.compaction?.auto === false ? 'manual' : 'auto'} window=${config.compaction?.contextWindowTokens ?? 128_000} skills=${ + agent + .listSkills() + .map((s) => s.name) + .join(', ') || '(none)' + }`, ) continue } @@ -249,6 +414,42 @@ export const runRepl = async (): Promise => { console.log('[history] cleared') continue } + if (prompt === '/compact') { + const r = await agent.compact({ history, force: true }) + if (r.compacted) { + history.splice(0, history.length, ...r.history) + console.log(`[compact] history: ~${r.beforeTokens} -> ~${r.afterTokens} tokens`) + } else { + console.log('[compact] nothing to compact') + } + continue + } + if (prompt === '/autopilot') { + const current = agent.getToolApprovalMode() + if (current === 'autopilot') { + agent.setToolApprovalMode(lastAskMode) + } else { + lastAskMode = current + agent.setToolApprovalMode('autopilot') + } + console.log(`[approval] mode=${agent.getToolApprovalMode()}`) + continue + } + if (prompt === '/approval' || prompt.startsWith('/approval ')) { + const mode = prompt.slice('/approval'.length).trim() + if (!mode) { + console.log(`[approval] mode=${agent.getToolApprovalMode()}`) + } else if (!APPROVAL_MODES.includes(mode as ToolApprovalMode)) { + console.log(`[approval] unknown mode "${mode}"; use ${APPROVAL_MODES.join(' | ')}`) + } else { + agent.setToolApprovalMode(mode as ToolApprovalMode) + if (mode !== 'autopilot') { + lastAskMode = mode as ToolApprovalMode + } + console.log(`[approval] mode=${mode}`) + } + continue + } if (prompt === '/reconnect') { try { await agent.reconnect() @@ -264,8 +465,7 @@ export const runRepl = async (): Promise => { continue } - streamingFinal = false - streamingThought = false + streaming = 'none' runController = new AbortController() try { const result = await agent.run({ @@ -274,15 +474,19 @@ export const runRepl = async (): Promise => { signal: runController.signal, }) sigIntCount = 0 + // The run compacted the history it was given: keep the compacted copy. + if (result.compactedHistory) { + history.splice(0, history.length, ...result.compactedHistory) + } history.push({ role: 'user', content: prompt }) history.push({ role: 'assistant', content: result.text }) - while (history.length > HISTORY_LIMIT) { - history.shift() - } + trimHistory(history) + const u = result.usage console.log( - `[meta] iterations=${result.iterations} steps=${result.trace.length} tokens=${result.usage.totalTokens} (in=${result.usage.inputTokens} out=${result.usage.outputTokens})`, + `[meta] iterations=${result.iterations} steps=${result.trace.length} tokens=${u.totalTokens} (in=${u.inputTokens} out=${u.outputTokens}${u.reasoningTokens ? ` reasoning=${u.reasoningTokens}` : ''}${u.cachedInputTokens ? ` cached=${u.cachedInputTokens}` : ''})`, ) } catch (err) { + endLine() if ((err as { name?: string })?.name === 'AbortError') { console.log('[abort] run cancelled') sigIntCount = 0 @@ -295,6 +499,7 @@ export const runRepl = async (): Promise => { } } finally { process.off('SIGINT', onSigInt) + rl.off('SIGINT', onSigInt) rl.close() // Library-side timeout caps both the (skipped here) wait for active runs // AND the MCP transport teardown, so close() can never hang the CLI. diff --git a/src/compaction.ts b/src/compaction.ts new file mode 100644 index 0000000..1c39ac5 --- /dev/null +++ b/src/compaction.ts @@ -0,0 +1,277 @@ +import { generateText, type LanguageModel, type Tool, type ToolSet } from 'ai' +import type { ICompactionConfig, IConversationTurn, IStepResult, IUsage } from './types.ts' +import { clipText, normalizeUsage, withTimeout } from './utils.ts' + +export const DEFAULT_CONTEXT_WINDOW_TOKENS = 128_000 +export const DEFAULT_MAX_TOOL_OUTPUT_CHARS = 20_000 + +export interface IResolvedCompaction { + auto: boolean + contextWindowTokens: number + thresholdTokens: number + keepRecentTurns: number + keepRecentSteps: number + summaryMaxTokens: number + maxToolOutputChars: number + clearToolResultsAfterTokens: number + keepToolResults: number +} + +const nonNeg = (v: number | undefined, fallback: number): number => + typeof v === 'number' && Number.isFinite(v) && v >= 0 ? Math.floor(v) : fallback + +export const resolveCompaction = (cfg: ICompactionConfig | undefined): IResolvedCompaction => { + const contextWindowTokens = nonNeg(cfg?.contextWindowTokens, 0) || DEFAULT_CONTEXT_WINDOW_TOKENS + return { + auto: cfg?.auto !== false, + contextWindowTokens, + thresholdTokens: nonNeg(cfg?.thresholdTokens, 0) || Math.floor(contextWindowTokens / 2), + keepRecentTurns: nonNeg(cfg?.keepRecentTurns, 4), + keepRecentSteps: nonNeg(cfg?.keepRecentSteps, 3), + summaryMaxTokens: nonNeg(cfg?.summaryMaxTokens, 0) || 1024, + maxToolOutputChars: nonNeg(cfg?.maxToolOutputChars, DEFAULT_MAX_TOOL_OUTPUT_CHARS), + clearToolResultsAfterTokens: nonNeg( + cfg?.clearToolResultsAfterTokens, + Math.floor(contextWindowTokens / 4), + ), + keepToolResults: nonNeg(cfg?.keepToolResults, 3), + } +} + +// Cheap, provider-agnostic token estimate: ~4 chars per token. +export const estimateTokens = (text: string): number => Math.ceil(text.length / 4) + +export const SUMMARY_PREFIX = '[Summary of earlier conversation]' + +export const isSummaryTurn = (t: IConversationTurn): boolean => + t.role === 'assistant' && t.content.startsWith(SUMMARY_PREFIX) + +export const renderTurns = (turns: IConversationTurn[]): string => + turns.map((t) => `${t.role}: ${t.content}`).join('\n') + +const HISTORY_SUMMARY_SYSTEM = `You compress conversation history for an AI agent. + +Summarize the conversation below into a compact, factual record that lets the agent continue the conversation without the original turns. Keep: what the user asked for and decided, facts and results the assistant reported (names, identifiers, numbers, URLs, file paths, dates), open questions and commitments. Drop greetings, filler and repetition. Write plain text in the conversation's language, no preamble.` + +const TRACE_SUMMARY_SYSTEM = `You compress the execution trace of an AI agent. + +Summarize the executed steps below into a compact, factual record the agent can rely on for the remaining steps. Keep every concrete result: identifiers, names, numbers, URLs, file paths, errors and what failed, and conclusions. Drop narration. Write plain text, no preamble.` + +export interface ICompactHistoryOptions { + // Compact only when the estimate exceeds this (default 64_000). + thresholdTokens?: number + // Turns kept verbatim (default 4). + keepRecentTurns?: number + // Output cap of the summary call (default 1024). + summaryMaxTokens?: number + // Compact regardless of the threshold. + force?: boolean + signal?: AbortSignal + timeoutMs?: number + // Usage of the summary call (it is NOT added to any run by itself). + onUsage?: (usage: IUsage) => void +} + +export interface ICompactHistoryResult { + history: IConversationTurn[] + summary?: string + beforeTokens: number + afterTokens: number + compacted: boolean + usage?: IUsage +} + +/** + * Summarise all but the last `keepRecentTurns` turns into ONE assistant turn + * prefixed "[Summary of earlier conversation]" (an earlier summary turn is + * folded into the new one). Never throws: on failure, or below the + * threshold without `force`, the input comes back unchanged. The input array + * is never mutated. + */ +export const compactHistory = async ( + turns: IConversationTurn[], + model: LanguageModel, + opts: ICompactHistoryOptions = {}, +): Promise => { + const beforeTokens = estimateTokens(renderTurns(turns)) + const unchanged: ICompactHistoryResult = { + history: turns, + beforeTokens, + afterTokens: beforeTokens, + compacted: false, + } + const threshold = opts.thresholdTokens ?? DEFAULT_CONTEXT_WINDOW_TOKENS / 2 + if (!opts.force && beforeTokens <= threshold) { + return unchanged + } + const keep = Math.max(0, opts.keepRecentTurns ?? 4) + const cut = turns.length - keep + if (cut <= 0) { + return unchanged + } + const older = turns.slice(0, cut) + if (older.length === 1 && isSummaryTurn(older[0])) { + return unchanged + } + try { + const r = await generateText({ + model, + instructions: HISTORY_SUMMARY_SYSTEM, + prompt: `Conversation to summarize:\n${renderTurns(older)}`, + maxOutputTokens: opts.summaryMaxTokens ?? 1024, + abortSignal: withTimeout(opts.signal, opts.timeoutMs ?? 0), + maxRetries: 0, + }) + const usage = normalizeUsage(r.usage) + opts.onUsage?.(usage) + const summary = r.text.trim() + if (!summary) { + return { ...unchanged, usage } + } + const history: IConversationTurn[] = [ + { role: 'assistant', content: `${SUMMARY_PREFIX}\n${summary}` }, + ...turns.slice(cut), + ] + return { + history, + summary, + beforeTokens, + afterTokens: estimateTokens(renderTurns(history)), + compacted: true, + usage, + } + } catch { + return unchanged + } +} + +export interface ISummarizeTraceOptions { + summaryMaxTokens?: number + signal?: AbortSignal + timeoutMs?: number + // Renders the steps (the prompt module's renderer, injected to avoid a + // circular import). + render: (steps: IStepResult[], offset: number) => string + // Index of steps[0] in the full trace. + offset: number +} + +/** + * Fold `steps` (and an earlier running summary) into a new running summary. + * Never throws: undefined on failure. + */ +export const summarizeTrace = async ( + steps: IStepResult[], + previousSummary: string | undefined, + model: LanguageModel, + opts: ISummarizeTraceOptions, +): Promise<{ summary: string; usage: IUsage } | undefined> => { + try { + const r = await generateText({ + model, + instructions: TRACE_SUMMARY_SYSTEM, + prompt: [ + previousSummary ? `Summary of the steps before these:\n${previousSummary}\n` : '', + `Steps to summarize:\n${opts.render(steps, opts.offset)}`, + ] + .filter(Boolean) + .join('\n'), + maxOutputTokens: opts.summaryMaxTokens ?? 1024, + abortSignal: withTimeout(opts.signal, opts.timeoutMs ?? 0), + maxRetries: 0, + }) + const summary = r.text.trim() + return summary ? { summary, usage: normalizeUsage(r.usage) } : undefined + } catch { + return undefined + } +} + +// ── model-visible tool output cap ──────────────────────────────────────── + +type ToolModelOutput = Awaited>> + +/** + * Clip what the MODEL sees of one tool result: text values, the JSON + * serialisation of json values (which then becomes text), and text parts of + * content values. Error / denied outputs pass through. Pure. + */ +export const truncateToolModelOutput = ( + output: ToolModelOutput, + maxChars: number, +): ToolModelOutput => { + if (!(maxChars > 0)) { + return output + } + switch (output.type) { + case 'text': + return output.value.length > maxChars + ? { ...output, value: clipText(output.value, maxChars) } + : output + case 'json': { + let s: string | undefined + try { + s = JSON.stringify(output.value) + } catch { + return output + } + return s !== undefined && s.length > maxChars + ? { + type: 'text', + value: clipText(s, maxChars), + ...(output.providerOptions ? { providerOptions: output.providerOptions } : {}), + } + : output + } + case 'content': + return { + ...output, + value: output.value.map((part) => + part.type === 'text' && part.text.length > maxChars + ? { ...part, text: clipText(part.text, maxChars) } + : part, + ), + } + default: + return output + } +} + +// What the AI SDK sends when a tool has no toModelOutput. +const defaultModelOutput = (output: unknown): ToolModelOutput => { + if (typeof output === 'string') { + return { type: 'text', value: output } + } + if (output === undefined) { + return { type: 'json', value: null } + } + try { + const s = JSON.stringify(output) + return { type: 'json', value: s === undefined ? null : JSON.parse(s) } + } catch { + return { type: 'text', value: String(output) } + } +} + +/** + * Wrap a tool so the model sees at most `maxChars` of each result (with a + * "… [truncated N chars]" marker). An existing toModelOutput runs first and + * its value is clipped. The raw output (trace, events) is untouched. + */ +export const withToolOutputLimit = (tool: T, maxChars: number): T => { + if (!(maxChars > 0)) { + return tool + } + const own = tool.toModelOutput + const toModelOutput: Tool['toModelOutput'] = async (opts) => + truncateToolModelOutput(own ? await own(opts) : defaultModelOutput(opts.output), maxChars) + return { ...tool, toModelOutput } +} + +export const withToolSetOutputLimit = (tools: ToolSet, maxChars: number): ToolSet => { + const out: ToolSet = {} + for (const [name, tool] of Object.entries(tools)) { + out[name] = withToolOutputLimit(tool, maxChars) + } + return out +} diff --git a/src/context-editing.ts b/src/context-editing.ts new file mode 100644 index 0000000..5b45226 --- /dev/null +++ b/src/context-editing.ts @@ -0,0 +1,95 @@ +import type { ModelMessage } from 'ai' +import { estimateTokens } from './compaction.ts' + +/** + * Context editing inside one step's tool loop — what Anthropic's + * `clear_tool_uses` and Claude Code's micro-compaction do: once the loop's + * conversation grows past `triggerTokens`, the OLDEST tool results are replaced + * with a one-line stub (the calls stay, so the model knows what it did and can + * call again), keeping the `keep` most recent results verbatim. Clearing is + * sticky — a result cleared once stays cleared — so the prefix the provider + * cached does not flip back and forth between rounds. + */ +export interface IToolResultClearing { + triggerTokens: number + // Most recent tool results kept verbatim (default 3). + keep?: number +} + +export interface IClearedInfo { + cleared: number + beforeTokens: number + afterTokens: number +} + +// A result's position in the conversation ("message:part"). The loop only +// appends, so positions are stable — unlike tool call ids, which some servers +// repeat across rounds. +const positionsOf = (messages: ModelMessage[]): string[] => + messages.flatMap((m, i) => + m.role === 'tool' && Array.isArray(m.content) + ? (m.content as { type: string }[]).flatMap((p, j) => + p.type === 'tool-result' ? [`${i}:${j}`] : [], + ) + : [], + ) + +const stub = (name: string): string => + `[${name} result cleared to save context — call the tool again if you still need it]` + +const estimate = (messages: ModelMessage[]): number => { + try { + return estimateTokens(JSON.stringify(messages)) + } catch { + return 0 + } +} + +// A stateful clearer for one loop: call it with each round's messages. +export const createToolResultClearer = ( + opts: IToolResultClearing, + onCleared?: (info: IClearedInfo) => void, +): ((messages: ModelMessage[]) => ModelMessage[]) => { + const keep = Math.max(0, opts.keep ?? 3) + const cleared = new Set() + + const apply = (messages: ModelMessage[]): ModelMessage[] => { + if (cleared.size === 0) { + return messages + } + return messages.map((m, i) => + m.role === 'tool' && Array.isArray(m.content) + ? { + ...m, + content: m.content.map((p, j) => + p.type === 'tool-result' && cleared.has(`${i}:${j}`) + ? { ...p, output: { type: 'text' as const, value: stub(p.toolName) } } + : p, + ), + } + : m, + ) + } + + return (messages) => { + const edited = apply(messages) + const before = estimate(edited) + if (before <= opts.triggerTokens) { + return edited + } + const positions = positionsOf(messages) + let added = 0 + for (const pos of positions.slice(0, Math.max(0, positions.length - keep))) { + if (!cleared.has(pos)) { + cleared.add(pos) + added += 1 + } + } + if (added === 0) { + return edited + } + const next = apply(messages) + onCleared?.({ cleared: added, beforeTokens: before, afterTokens: estimate(next) }) + return next + } +} diff --git a/src/context.ts b/src/context.ts index 43d72ee..4fdf328 100644 --- a/src/context.ts +++ b/src/context.ts @@ -1,4 +1,6 @@ import { AsyncLocalStorage } from 'node:async_hooks' +import type { IRunState } from './internal.ts' +import type { EventHandler, IPlanStep } from './types.ts' export interface IRunContext { runId: string @@ -7,6 +9,15 @@ export interface IRunContext { // first tool that needs to spill binary content; consumers can probe it // via getCurrentRunSandbox() and write into it directly. sandboxDir: string + // The run's event sink (tags runId, accumulates usage). Set by the runner + // once the run starts; tool wrappers (approval gate, find_tools, skills, + // subagents) emit through it so their events land in the right run. + emit?: EventHandler + // The plan step currently executing, for events emitted from inside tools. + currentStep?: IPlanStep + // Mutable per-run state shared with the tool wrappers (usage, tool-call + // count, discovered tools, active skills). + state?: IRunState } // Per-run context. Propagates through awaits, so any code reachable from diff --git a/src/executor.ts b/src/executor.ts index c7b6d42..43d602d 100644 --- a/src/executor.ts +++ b/src/executor.ts @@ -1,12 +1,23 @@ -import { stepCountIs, streamText, type ToolSet } from 'ai' +import { stepCountIs, streamText, type ModelMessage, type StopCondition, type ToolSet } from 'ai' +import { resolvePromptCaching, withRollingBreakpoint } from './caching.ts' +import { resolveCompaction } from './compaction.ts' +import { createToolResultClearer } from './context-editing.ts' +import { stageCallOptions } from './call-options.ts' import type { IAgentInternalContext } from './internal.ts' -import { EXECUTOR_SYSTEM, buildExecutorUserPrompt, withDomainContext } from './prompts.ts' +import { hasTokenLimits, limitsStopCondition, resolveLimits } from './limits.ts' +import { effectiveToolStrategy } from './planner.ts' +import { buildExecutorUserPrompt, composeExecutorSystem } from './prompts.ts' +import { stageEmitsThoughts } from './thinking.ts' +import { MAX_CARRIED_DISCOVERED_TOOLS } from './tool-search.ts' import { ATTR, withSpan } from './tracing.ts' import type { IConversationTurn, IPlan, IPlanStep, IStepResult, IUsage } from './types.ts' -import { withRetry, withTimeout } from './utils.ts' +import { emptyUsage, normalizeUsage, withRetry, withTimeout } from './utils.ts' +// The host + MCP tools a step may use under 'all' / 'plan-narrowed' (the +// 'search' strategy builds its active set per LLM step instead; built-in +// tools are added on top by the executor). export const buildActiveToolSet = (ctx: IAgentInternalContext, step: IPlanStep): ToolSet => { - if (ctx.config.toolSelectionStrategy !== 'plan-narrowed') { + if (effectiveToolStrategy(ctx) !== 'plan-narrowed') { return ctx.tools } const allowed = new Set(step.suggestedTools ?? []) @@ -103,6 +114,43 @@ export const executeStep = async ( }, ) +// The tool set with its keys in name order: the tool list heads every cached +// prompt prefix, so it must be identical however the MCP servers connected. +export const sortTools = >(tools: T): T => + Object.fromEntries(Object.entries(tools).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) as T + +// Tools for one executor call: what the strategy exposes + the built-ins. +// In 'search' mode the full catalogue is passed and `prepareStep` narrows +// each LLM step to the active set (find_tools grows it mid-call). +export const buildExecutorTools = ( + ctx: IAgentInternalContext, + step: IPlanStep, +): { tools: ToolSet; activeTools?: string[]; active?: Set } => { + const strategy = effectiveToolStrategy(ctx) + const builtins = ctx.builtinTools ?? {} + if (strategy === 'search') { + const tools: ToolSet = { ...ctx.tools, ...builtins, ...(ctx.findTools ?? {}) } + const active = new Set([...Object.keys(builtins), ...Object.keys(ctx.findTools ?? {})]) + for (const name of step.suggestedTools ?? []) { + if (ctx.tools[name]) { + active.add(name) + } + } + for (const name of (ctx.run?.discovered ?? []).slice(-MAX_CARRIED_DISCOVERED_TOOLS)) { + if (ctx.tools[name]) { + active.add(name) + } + } + return { tools: sortTools(tools), active } + } + const base = buildActiveToolSet(ctx, step) + const tools: ToolSet = sortTools(Object.keys(builtins).length ? { ...base, ...builtins } : base) + // Defence-in-depth: explicitly tell the SDK which tools are callable in + // this step. Only worth it in narrowed mode; in 'all' mode it's just a + // copy of every key, equivalent to omitting the field. + return strategy === 'plan-narrowed' ? { tools, activeTools: Object.keys(tools) } : { tools } +} + const runOnce = async ( input: string, plan: IPlan, @@ -115,8 +163,11 @@ const runOnce = async ( const toolCalls: IStepResult['toolCalls'] = [] const toolInputs = new Map() - const activeTools = buildActiveToolSet(ctx, step) - const narrowed = ctx.config.toolSelectionStrategy === 'plan-narrowed' + const { tools, activeTools, active } = buildExecutorTools(ctx, step) + const run = ctx.run + if (run) { + run.stepActive = active + } const sanitizeForEvent = async (toolName: string, raw: unknown): Promise => { const fn = ctx.config.inputSanitizer @@ -140,16 +191,95 @@ const runOnce = async ( } } + // Stop conditions: the per-step LLM-step cap, plus the run-level token / + // tool-call budgets so a runaway tool loop stops at the next step boundary + // instead of at the end of the step. + let stoppedBy: 'tokens' | 'tool-calls' | undefined + const stopWhen: StopCondition[] = [stepCountIs(ctx.config.maxStepsPerTask)] + const limits = resolveLimits(ctx.config) + if (run && hasTokenLimits(limits)) { + stopWhen.push( + limitsStopCondition( + () => run.usage, + limits, + () => { + stoppedBy = 'tokens' + }, + ), + ) + } + const maxToolCalls = ctx.config.maxToolCalls + if (run && typeof maxToolCalls === 'number' && maxToolCalls > 0) { + stopWhen.push(() => { + if (run.toolCalls >= maxToolCalls) { + stoppedBy ??= 'tool-calls' + return true + } + return false + }) + } + + const call = stageCallOptions( + ctx.config, + 'executor', + composeExecutorSystem({ + domain: ctx.config.systemPrompt, + skills: ctx.skills, + activeSkills: run?.activeSkills, + searchMode: Boolean(active), + toolCount: ctx.toolCatalog.length, + }), + ) + const emitThoughts = stageEmitsThoughts(ctx.config, 'executor') + const view = run ? { summary: run.traceSummary, upTo: run.traceSummaryUpTo } : undefined + + // Before every round: re-read the search-mode active set (find_tools grows + // it), clear stale tool results past the threshold, and roll the cache + // breakpoint to the newest message. + const caching = resolvePromptCaching(ctx.config.promptCaching) + const compaction = resolveCompaction(ctx.config.compaction) + const clear = + compaction.clearToolResultsAfterTokens > 0 + ? createToolResultClearer( + { + triggerTokens: compaction.clearToolResultsAfterTokens, + keep: compaction.keepToolResults, + }, + (info) => { + ctx.emit({ + type: 'log', + level: 'info', + message: `[executor] cleared ${info.cleared} stale tool result(s) from the step's context`, + }) + ctx.emit({ + type: 'context.compacted', + scope: 'tool-results', + beforeTokens: info.beforeTokens, + afterTokens: info.afterTokens, + }) + }, + ) + : undefined + const editMessages = caching.enabled || Boolean(clear) + const result = streamText({ model: ctx.executorModel, - tools: activeTools, - // Defence-in-depth: explicitly tell the SDK which tools are callable in - // this step. Only worth it in narrowed mode; in 'all' mode it's just a - // copy of every key, equivalent to omitting the field. - ...(narrowed ? { activeTools: Object.keys(activeTools) } : {}), - stopWhen: stepCountIs(ctx.config.maxStepsPerTask), - system: withDomainContext(EXECUTOR_SYSTEM, ctx.config.systemPrompt), - prompt: buildExecutorUserPrompt(input, plan, step, trace, history), + tools, + ...(activeTools ? { activeTools } : {}), + ...(active || editMessages + ? { + prepareStep: ({ messages }: { messages: ModelMessage[] }) => { + const next = withRollingBreakpoint(clear ? clear(messages) : messages, caching) + return { + ...(active ? { activeTools: [...active].filter((name) => name in tools) } : {}), + ...(editMessages ? { messages: next } : {}), + } + }, + } + : {}), + stopWhen, + ...call, + prompt: buildExecutorUserPrompt(input, plan, step, trace, history, view), abortSignal: withTimeout(signal, ctx.config.llmTimeoutMs ?? 0), }) @@ -160,6 +290,11 @@ const runOnce = async ( ctx.emit({ type: 'step.text-delta', step, delta: part.text }) } break + case 'reasoning-delta': + if (part.text && emitThoughts) { + ctx.emit({ type: 'step.reasoning-delta', step, delta: part.text }) + } + break case 'tool-call': { const sanitizedInput = await sanitizeForEvent(part.toolName, part.input) toolInputs.set(part.toolCallId, { name: part.toolName, input: sanitizedInput }) @@ -167,6 +302,10 @@ const runOnce = async ( break } case 'tool-result': { + // A streaming tool's intermediate values; only the final one counts. + if ((part as { preliminary?: boolean }).preliminary) { + break + } const known = toolInputs.get(part.toolCallId) // Fallback path is defensive (tool-result before tool-call should // not happen). Sanitize the raw fallback input so an out-of-order @@ -219,18 +358,16 @@ const runOnce = async ( const summary = cleaned.length > 0 ? truncateForTrace(cleaned) - : toolCalls.length > 0 - ? `Executed ${toolCalls.length} tool call(s) without producing a final message; consider raising maxStepsPerTask.` - : 'Step produced no output.' + : stoppedBy + ? `Stopped early: the run's ${stoppedBy === 'tokens' ? 'token' : 'tool-call'} budget was reached after ${toolCalls.length} tool call(s).` + : toolCalls.length > 0 + ? `Executed ${toolCalls.length} tool call(s) without producing a final message; consider raising maxStepsPerTask.` + : 'Step produced no output.' return { summary, toolCalls, blocked, - usage: { - inputTokens: usage.inputTokens ?? 0, - outputTokens: usage.outputTokens ?? 0, - totalTokens: usage.totalTokens ?? 0, - }, + usage: usage ? normalizeUsage(usage) : emptyUsage(), } } diff --git a/src/index.ts b/src/index.ts index f73d2fa..fe625e8 100644 --- a/src/index.ts +++ b/src/index.ts @@ -1,19 +1,7 @@ -import type { LanguageModel } from 'ai' -import type { IAgentInternalContext } from './internal.ts' -import { connectMcpServers, filterTools } from './mcp.ts' -import { buildModelFromStage, resolveStage } from './provider.ts' -import { runAgentLoop } from './runner.ts' -import type { - AgentEvent, - EventHandler, - IAgentConfig, - IAgentRunOptions, - IAgentRunResult, - LogLevel, -} from './types.ts' - +export { createAgent } from './agent.ts' +export type { IAgent, ICloseOptions, ICompactOptions, ICompactResult } from './agent.ts' export { getCurrentRunId, getCurrentRunSandbox } from './context.ts' -export { redactHeaders } from './utils.ts' +export { addUsage, normalizeUsage, redactHeaders } from './utils.ts' export { beginMcpOAuth, createNodeOAuthProvider, @@ -24,458 +12,83 @@ export { } from './mcp-oauth.ts' export type { INodeOAuthProviderOptions, IOAuthRequestOptions, IOAuthStore } from './mcp-oauth.ts' +// Thinking +export { mergeProviderOptions, resolveThinking } from './thinking.ts' +export type { IResolvedThinking } from './thinking.ts' +// Token limits +export { checkLimits, resolveLimits } from './limits.ts' +// Compaction +export { + compactHistory, + estimateTokens, + SUMMARY_PREFIX, + truncateToolModelOutput, + withToolOutputLimit, +} from './compaction.ts' +export type { ICompactHistoryOptions, ICompactHistoryResult } from './compaction.ts' +// Skills +export { defineSkill, loadSkillsFromDir, parseSkillMarkdown } from './skills.ts' +// Tool approval +export { + decideToolPermission, + isReadOnlyTool, + markReadOnly, + matchToolRule, + ToolDeniedError, +} from './approval.ts' +export type { IToolDecision, IToolDecisionInput } from './approval.ts' +export { ToolBudgetError } from './tool-wrap.ts' +// Tool search +export { searchTools } from './tool-search.ts' +export type { ISearchToolsOptions } from './tool-search.ts' +// Prompt caching +export { buildInstructions, resolvePromptCaching, withRollingBreakpoint } from './caching.ts' +// Context editing (stale tool results inside a step's tool loop) +export { createToolResultClearer } from './context-editing.ts' +export type { IClearedInfo, IToolResultClearing } from './context-editing.ts' +// Subagents +export { createSubagentTool } from './subagent.ts' +export type { ISubagentToolOptions, SubagentIsolation } from './subagent.ts' + export type { AgentEvent, + AgentStage, + BudgetKind, EventHandler, IAgentConfig, IAgentRunOptions, IAgentRunResult, IAgentStageOverride, + ICompactionConfig, IConversationTurn, + ILimitBreach, IMcpHttpServerConfig, IMcpServerConfig, IMcpStdioServerConfig, IPersistence, IRunSnapshot, + ISkill, IStepStartInfo, IPlan, IPlanStep, IStepResult, + IThinkingConfig, + ITokenLimits, + IToolApprovalConfig, + IToolApprovalRequest, + IToolCatalogEntry, IUsage, LogLevel, + PromptCachingSetting, ProviderType, ReplanCause, ReplanTrigger, + ThinkingLevel, + ThinkingSetting, + TokenLimitKind, + ToolApprovalDecision, + ToolApprovalMode, + ToolPermission, ToolSelectionStrategy, + UsagePhase, } from './types.ts' - -export interface ICloseOptions { - // When true, close() polls until activeRuns reaches 0 (or timeoutMs elapses) - // before tearing down MCP connections. Default false: close immediately and - // let active runs fail mid-flight (the legacy behavior). - waitForRuns?: boolean - // Cap, in ms, applied to the entire close() call: - // 1. when waitForRuns is true: max time spent waiting for active runs to - // drain; - // 2. ALWAYS: max time spent on the underlying MCP transport teardown - // (some HTTP/SSE transports can hang on close if the peer is - // unresponsive). After this elapses close() resolves anyway and the - // transport is abandoned to GC. Default 30s. - timeoutMs?: number -} - -export interface IAgent { - // Multiple concurrent runs are supported on a single agent instance: each - // call gets its own runId via AsyncLocalStorage, its own usage accumulator, - // its own abort signal, and its own onEvent. Tools and models are shared. - // Reconnect throws if runs are in flight; close optionally waits. - run: (options: IAgentRunOptions) => Promise - listTools: () => { name: string; description: string }[] - // Drops the current MCP connections and reconnects with fresh headers - // (via getHeaders if configured). Throws if any runs are in progress. - reconnect: () => Promise - close: (options?: ICloseOptions) => Promise - activeRuns: () => number -} - -const LEVEL_RANK: Record = { - none: 0, - error: 1, - warn: 2, - info: 3, - debug: 4, -} - -export const createAgent = async ( - config: IAgentConfig, - baseEventHandler?: EventHandler, -): Promise => { - const debugMode = LEVEL_RANK[config.logLevel] >= LEVEL_RANK.debug - - const emit: EventHandler = (event: AgentEvent) => { - if (baseEventHandler) { - try { - baseEventHandler(event) - } catch (err) { - if (debugMode) { - // Surface handler errors only in debug mode; in normal mode they - // are silently swallowed to keep the agent robust to bad consumers. - // Write directly to console to avoid recursion via emit(). - console.error('[agent] event handler threw:', err) - } - } - } - } - - const log = (level: 'info' | 'warn' | 'error', message: string) => { - if (LEVEL_RANK[config.logLevel] >= LEVEL_RANK[level]) { - emit({ type: 'log', level, message }) - } - } - - // Forward declaration: connectMcpServers needs the onToolsChanged callback, - // and the callback must enqueue refreshes that drain only when activeRuns - // reaches zero. We set the impl after the run/close machinery is wired up. - let onToolsChanged: ((server: string) => void) | undefined - // Merge native tools (config.tools) into a freshly filtered MCP view. - // Called both at startup and after a tools/list_changed refresh, so native - // entries survive MCP catalog rebuilds. - const mergeNativeTools = ( - f: { - tools: ReturnType['tools'] - catalog: ReturnType['catalog'] - }, - onCollision: (name: string) => never, - ): { - tools: ReturnType['tools'] - catalog: ReturnType['catalog'] - } => { - if (!config.tools) { - return f - } - const nativeCatalog: ReturnType['catalog'] = [] - for (const [name, tool] of Object.entries(config.tools)) { - if (f.tools[name]) { - onCollision(name) - } - // availableTools wins over native registration, mirroring the MCP - // path: an explicit allowlist excludes everything not on it. - if (config.availableTools?.length && !config.availableTools.includes(name)) { - continue - } - if ( - !config.availableTools?.length && - config.excludedTools?.length && - config.excludedTools.includes(name) - ) { - continue - } - f.tools[name] = tool - const rawDesc = - typeof tool === 'object' && tool && 'description' in tool - ? (tool as { description?: unknown }).description - : '' - nativeCatalog.push({ - name, - description: typeof rawDesc === 'string' ? rawDesc : '', - server: '', - }) - } - return { tools: f.tools, catalog: [...f.catalog, ...nativeCatalog] } - } - - const connect = async () => { - const c = await connectMcpServers( - config.mcpServers, - log, - config.clientName, - config.outputSanitizer, - (server) => onToolsChanged?.(server), - config.inputSanitizer, - ) - const f = filterTools(c.tools, c.catalog, config.availableTools, config.excludedTools, log) - const merged = mergeNativeTools(f, (name) => { - // Tear the connection down so we don't leak open MCP transports when - // the caller's misconfiguration crashes startup. - void c.close().catch(() => {}) - throw new Error(`Native tool "${name}" collides with an MCP-registered tool of the same name`) - }) - return { connection: c, tools: merged.tools, catalog: merged.catalog } - } - - let connected: Awaited> - try { - connected = await connect() - } catch (err) { - emit({ - type: 'error', - error: err instanceof Error ? err : new Error(String(err)), - phase: 'init', - }) - throw err - } - - // Hard fail when explicitly requested and every configured server failed - // to connect. Without this, the agent would start with zero tools, the - // planner would produce a no-tool plan, and the user only sees the - // problem several seconds later when execution times out. We only - // enforce this when servers were configured at all - an agent that runs - // tool-less by design (mcpServers: {}) is a valid use case. - const configuredServers = Object.keys(config.mcpServers).length - if (config.failOnNoTools && configuredServers > 0) { - const anyConnected = connected.connection.results.some((r) => r.connected) - if (!anyConnected) { - const reasons = connected.connection.results - .filter((r) => !r.connected) - .map((r) => `${r.name}: ${r.error ?? 'unknown'}`) - .join('; ') - await connected.connection.close().catch(() => {}) - const err = new Error( - `All ${configuredServers} configured MCP server(s) failed to connect [${reasons}]`, - ) - emit({ type: 'error', error: err, phase: 'init' }) - throw err - } - } - - // Executor always inherits the top-level config; planner / synthesizer can - // override any field (provider, baseURL, apiKey, model) via the dedicated - // override blocks. Legacy plannerModel / synthesizerModel still work as - // model-only shortcuts when the override block is absent. - // - // Provider SDKs are dynamically imported (peerDependencies, optional), so - // model construction is async; build all three in parallel since they are - // independent. On failure (e.g. missing peer dep) we MUST close the MCP - // connection opened above - otherwise we leak open transports for what is - // typically a misconfiguration retry-loop. - let executorModel: LanguageModel - let plannerModel: LanguageModel - let synthesizerModel: LanguageModel - try { - const executorStage = resolveStage(config, undefined, undefined, 'executor') - const plannerStage = resolveStage(config, config.planner, config.plannerModel, 'planner') - const synthesizerStage = resolveStage( - config, - config.synthesizer, - config.synthesizerModel, - 'synthesizer', - ) - ;[executorModel, plannerModel, synthesizerModel] = await Promise.all([ - buildModelFromStage(config.clientName, executorStage), - buildModelFromStage(config.clientName, plannerStage), - buildModelFromStage(config.clientName, synthesizerStage), - ]) - } catch (err) { - await connected.connection.close().catch(() => {}) - emit({ - type: 'error', - error: err instanceof Error ? err : new Error(String(err)), - phase: 'init', - }) - throw err - } - - const ctx: IAgentInternalContext = { - config, - executorModel, - plannerModel, - synthesizerModel, - tools: connected.tools, - toolCatalog: connected.catalog.map((c) => ({ name: c.name, description: c.description })), - emit, - } - - let activeRuns = 0 - let closed = false - let reconnecting = false - let refreshing = false - // Server names whose tools/list_changed has fired but whose refresh is - // deferred because activeRuns > 0 or another lifecycle op is in flight. - const pendingRefreshes = new Set() - - const applyRefresh = async (server: string): Promise => { - await connected.connection.refreshServer(server) - // Re-apply the availableTools/excludedTools filter to the freshly mutated - // raw maps, then mutate ctx.tools in place so existing closures pick up - // the new set. The synchronous delete+assign block runs without awaits, - // so no run() can interleave once activeRuns has been gated to 0. - const f = filterTools( - connected.connection.tools, - connected.connection.catalog, - config.availableTools, - config.excludedTools, - ) - // Native tools must be merged back in: filterTools/refreshServer only - // produce the MCP view, so without this they'd vanish from ctx.tools - // until the next reconnect. - const merged = mergeNativeTools(f, (name) => { - throw new Error(`Native tool "${name}" collides with an MCP-registered tool of the same name`) - }) - for (const k of Object.keys(ctx.tools)) { - delete ctx.tools[k] - } - Object.assign(ctx.tools, merged.tools) - ctx.toolCatalog = merged.catalog.map((c) => ({ name: c.name, description: c.description })) - log( - 'info', - `[mcp] ${server}: tool list refreshed (${merged.catalog.length} tools total after filter)`, - ) - } - - const drainPendingRefreshes = async (): Promise => { - if (refreshing || closed || reconnecting) { - return - } - if (activeRuns > 0 || pendingRefreshes.size === 0) { - return - } - refreshing = true - try { - while (pendingRefreshes.size > 0 && activeRuns === 0 && !closed && !reconnecting) { - const next = pendingRefreshes.values().next().value as string - pendingRefreshes.delete(next) - try { - await applyRefresh(next) - } catch (err) { - log('warn', `[mcp] ${next}: refresh failed - ${(err as Error).message}`) - } - } - } finally { - refreshing = false - } - } - - // Wired up here so the closure captures the lifecycle flags and - // pendingRefreshes set declared above. - onToolsChanged = (server: string) => { - pendingRefreshes.add(server) - // Sync entry into drainPendingRefreshes runs to the first await before - // returning, so the refreshing/activeRuns gating is observed atomically - // from any run() call that lands after this notification. - void drainPendingRefreshes() - } - - return { - run: async (options: IAgentRunOptions) => { - if (closed) { - throw new Error('Agent is closed') - } - // Block new runs from racing into a mid-flight reconnect: the await on - // connect() inside reconnect() opens a window during which ctx.tools is - // about to be mutated. Accepting new runs in that window would let them - // observe a half-deleted ToolSet. - if (reconnecting) { - throw new Error('Agent is reconnecting; retry shortly') - } - if (refreshing) { - throw new Error('Agent is refreshing tools; retry shortly') - } - const cap = config.maxConcurrentRuns - if (typeof cap === 'number' && cap > 0 && activeRuns >= cap) { - // Synchronous reject: the caller should rate-limit on its side; we - // intentionally avoid a queue so close()/reconnect() stay simple - // (no pending-promises bookkeeping to drain). - throw new Error( - `maxConcurrentRuns reached (${activeRuns}/${cap}); rate-limit on the caller side or raise the cap`, - ) - } - activeRuns++ - try { - return await runAgentLoop(ctx, options) - } finally { - activeRuns-- - // Drain deferred tool refreshes once we go quiescent. Fire-and-forget: - // the next caller of run() either sees the refresh applied or gets - // the 'refreshing' rejection and retries. - void drainPendingRefreshes() - } - }, - listTools: () => ctx.toolCatalog.slice(), - reconnect: async () => { - if (closed) { - throw new Error('Agent is closed') - } - if (reconnecting) { - throw new Error('Already reconnecting') - } - if (refreshing) { - throw new Error('Agent is refreshing tools; retry shortly') - } - if (activeRuns > 0) { - throw new Error(`Cannot reconnect with ${activeRuns} active run(s)`) - } - reconnecting = true - try { - const oldConnection = connected.connection - // Drop any deferred per-server refreshes - the new connection comes - // up with a fresh tool listing and its own subscriptions, so the - // pending entries are stale and would only re-trigger work. - pendingRefreshes.clear() - const fresh = await connect() - // Mutate ctx.tools in place so existing closures pick up new tools. - // The reconnecting flag held above blocks new run() calls during the - // await + mutation window, so no concurrent reader sees inconsistent - // state. - for (const k of Object.keys(ctx.tools)) { - delete ctx.tools[k] - } - Object.assign(ctx.tools, fresh.tools) - ctx.toolCatalog = fresh.catalog.map((c) => ({ name: c.name, description: c.description })) - connected = fresh - // Close old connections only after the new ones are wired up so a - // failed reconnect doesn't leave the agent without tools. - await oldConnection.close().catch(() => {}) - } finally { - reconnecting = false - } - }, - close: async (options?: ICloseOptions) => { - closed = true - const waitForRuns = options?.waitForRuns ?? false - const timeoutMs = options?.timeoutMs ?? 30_000 - // Single budget shared by both phases: waiting for runs to drain (if - // requested) and the MCP transport teardown. Whatever waitForRuns - // consumes is subtracted from what's available to the close itself, - // with a small floor so the transport always gets *some* chance. - const deadline = Date.now() + timeoutMs - if (waitForRuns && activeRuns > 0) { - // Poll instead of using EventEmitter to avoid coupling close() to - // event-handler ordering; activeRuns flips back to 0 after the run's - // try/finally regardless. - while (activeRuns > 0 && Date.now() < deadline) { - await new Promise((r) => setTimeout(r, 50)) - } - } - if (activeRuns > 0) { - emit({ - type: 'log', - level: 'warn', - message: `[agent] close called with ${activeRuns} active run(s); they will fail mid-flight`, - }) - } - // Race the MCP teardown against the remaining budget. An unresponsive - // HTTP/SSE peer can leave client.close() pending forever; the timeout - // guarantees close() itself resolves so callers (CLIs, test fixtures) - // never hang. Transports abandoned this way are GC'd when the agent is. - // - // Two tripwires we MUST get right: - // 1. clearTimeout when MCP wins the race - otherwise the timer keeps - // the event loop alive for the FULL closeBudget after close() - // returned, and processes that called close() at the end of main - // visibly hang. Verified manually: with 5s budget the process - // exited 5s late before this clear. - // 2. timeoutId.unref() - belt and suspenders so even an exception in - // the race body cannot pin the loop. - // - // No artificial floor: timeoutMs is a hard cap on the whole close() - // call (per ICloseOptions docs). If waitForRuns already burned the - // budget, we hand the teardown a 0ms slot - setTimeout(_, 0) still - // schedules to the next tick, so transports that finish synchronously - // can still win, but nothing here will exceed timeoutMs. - const closeBudget = Math.max(deadline - Date.now(), 0) - let timedOut = false - let timeoutId: ReturnType | undefined - await Promise.race([ - connected.connection.close().catch(() => {}), - new Promise((resolve) => { - timeoutId = setTimeout(() => { - timedOut = true - resolve() - }, closeBudget) - timeoutId.unref?.() - }), - ]) - if (timeoutId) { - clearTimeout(timeoutId) - } - if (timedOut) { - emit({ - type: 'log', - level: 'warn', - message: `[agent] MCP teardown exceeded ${closeBudget}ms; abandoning transport(s)`, - }) - } - }, - activeRuns: () => activeRuns, - } -} diff --git a/src/internal.ts b/src/internal.ts index 7dd13ef..4ff55b3 100644 --- a/src/internal.ts +++ b/src/internal.ts @@ -1,12 +1,46 @@ import type { LanguageModel, ToolSet } from 'ai' -import type { EventHandler, IAgentConfig } from './types.ts' +import type { IApprovalController } from './approval.ts' +import type { EventHandler, IAgentConfig, ISkill, IToolCatalogEntry, IUsage } from './types.ts' + +// The effective (resolved) tool selection strategy of a run. +export type EffectiveToolStrategy = 'all' | 'plan-narrowed' | 'search' + +// Mutable state of ONE run, shared between the runner, the stages and the +// tool wrappers (reached from inside a tool through the run context). +export interface IRunState { + // Live usage accumulator of the run (the object the result returns). + usage: IUsage + // Tool calls made so far (maxToolCalls). + toolCalls: number + strategy: EffectiveToolStrategy + // Tools find_tools activated, oldest first (search strategy). + discovered: string[] + // Tools active in the executor call in flight (search strategy); find_tools + // adds to it so the next LLM step sees the new tools. + stepActive?: Set + // Skill names active for the rest of the run, in activation order. + activeSkills: string[] + // Running summary of trace[0, traceSummaryUpTo). + traceSummary?: string + traceSummaryUpTo: number +} export interface IAgentInternalContext { config: IAgentConfig executorModel: LanguageModel plannerModel: LanguageModel synthesizerModel: LanguageModel + // Host + MCP tools, already wrapped (approval gate, tool-call budget, + // model-visible output cap). Mutated in place on refresh / reconnect. tools: ToolSet - toolCatalog: { name: string; description: string }[] + toolCatalog: IToolCatalogEntry[] emit: EventHandler + // Built-in skill tools (load_skill / read_skill_file), when skills exist. + builtinTools?: ToolSet + // The find_tools built-in, when the search strategy is possible. + findTools?: ToolSet + skills?: ISkill[] + approval?: IApprovalController + // Per-run state; set by the runner on its per-run copy of the context. + run?: IRunState } diff --git a/src/limits.ts b/src/limits.ts new file mode 100644 index 0000000..ab1f4be --- /dev/null +++ b/src/limits.ts @@ -0,0 +1,78 @@ +import type { StopCondition, ToolSet } from 'ai' +import type { IAgentConfig, ILimitBreach, ITokenLimits, IUsage, TokenLimitKind } from './types.ts' +import { addUsage, emptyUsage, normalizeUsage, type ISdkUsageLike } from './utils.ts' + +// The effective limits of a run: `limits` with the legacy top-level +// maxTotalTokens folded in (limits.maxTotalTokens wins when both are set). +export const resolveLimits = ( + config: Pick, +): ITokenLimits => { + const maxTotalTokens = config.limits?.maxTotalTokens ?? config.maxTotalTokens + return { + ...config.limits, + ...(maxTotalTokens !== undefined ? { maxTotalTokens } : {}), + } +} + +const isCap = (cap: number | undefined): cap is number => + typeof cap === 'number' && Number.isFinite(cap) && cap > 0 + +export const hasTokenLimits = (limits: ITokenLimits | undefined): boolean => + Boolean( + limits && + (isCap(limits.maxInputTokens) || + isCap(limits.maxOutputTokens) || + isCap(limits.maxReasoningTokens) || + isCap(limits.maxTotalTokens)), + ) + +/** + * The first cumulative cap `usage` has reached (>=), or undefined. Pure. + * Checked in order input, output, reasoning, total; a cap of 0 / undefined + * is "no cap". + */ +export const checkLimits = ( + usage: IUsage, + limits: ITokenLimits | undefined, +): ILimitBreach | undefined => { + if (!limits) { + return undefined + } + const checks: [TokenLimitKind, number, number | undefined][] = [ + ['input', usage.inputTokens, limits.maxInputTokens], + ['output', usage.outputTokens, limits.maxOutputTokens], + ['reasoning', usage.reasoningTokens ?? 0, limits.maxReasoningTokens], + ['total', usage.totalTokens, limits.maxTotalTokens], + ] + for (const [kind, tokens, cap] of checks) { + if (isCap(cap) && tokens >= cap) { + return { kind, tokens, cap } + } + } + return undefined +} + +// Sum the usage of the steps of an in-flight streamText call. +export const sumStepUsage = (steps: { usage?: ISdkUsageLike }[]): IUsage => + steps.reduce((acc, s) => addUsage(acc, normalizeUsage(s.usage)), emptyUsage()) + +/** + * A stopWhen condition for the executor's tool loop: stops at the next step + * boundary once the run's usage so far plus this call's steps cross a cap, + * so a runaway tool loop cannot burn through the budget inside one step. + * `runUsage` is read live (subagent usage lands in it mid-step). + */ +export const limitsStopCondition = + ( + runUsage: () => IUsage, + limits: ITokenLimits, + onStop?: (breach: ILimitBreach) => void, + ): StopCondition => + ({ steps }) => { + const breach = checkLimits(addUsage(runUsage(), sumStepUsage(steps)), limits) + if (breach) { + onStop?.(breach) + return true + } + return false + } diff --git a/src/mcp.ts b/src/mcp.ts index 9e26b71..ac001d1 100644 --- a/src/mcp.ts +++ b/src/mcp.ts @@ -17,6 +17,7 @@ import { mkdir, writeFile } from 'node:fs/promises' import path from 'node:path' import { URL } from 'node:url' import packageJson from '../package.json' with { type: 'json' } +import { markReadOnly } from './approval.ts' import { getCurrentRunSandbox } from './context.ts' import { assertSecureOAuthUrl } from './mcp-oauth.ts' import { ATTR, withSpan } from './tracing.ts' @@ -83,6 +84,38 @@ interface IMcpToolDescriptor { name: string description?: string inputSchema: unknown + annotations?: { readOnlyHint?: boolean } +} + +// Default connect + tools/list budget per server. +export const DEFAULT_MCP_CONNECT_TIMEOUT_MS = 30_000 +// tools/list pagination cap: a server returning a cursor forever must not +// hang the connect. +export const MAX_TOOL_LIST_PAGES = 100 + +/** + * Every tool of a server, following `nextCursor` (up to MAX_TOOL_LIST_PAGES + * pages; a repeated cursor ends the walk). + */ +export const listAllTools = async ( + client: Pick, + onTruncated?: () => void, +): Promise => { + const all: IMcpToolDescriptor[] = [] + let cursor: string | undefined + const seen = new Set() + for (let page = 0; page < MAX_TOOL_LIST_PAGES; page++) { + const r = await client.listTools(cursor ? { cursor } : undefined) + all.push(...(r.tools as IMcpToolDescriptor[])) + const next = typeof r.nextCursor === 'string' && r.nextCursor ? r.nextCursor : undefined + if (!next || seen.has(next)) { + return all + } + seen.add(next) + cursor = next + } + onTruncated?.() + return all } interface IServerConnect { @@ -186,7 +219,7 @@ export interface IConnectedMcp { // in place. Callers should always read through these references rather than // caching their own snapshot. tools: ToolSet - catalog: { name: string; description: string; server: string }[] + catalog: { name: string; description: string; server: string; readOnly?: boolean }[] close: () => Promise // Per-server connect outcome so the caller can decide whether to fail hard // (e.g. when every configured server failed and the agent would otherwise @@ -213,6 +246,11 @@ export const connectMcpServers = async ( // already sanitized them for event emission - idempotency is documented // and required). inputSanitizer?: (toolName: string, input: unknown) => unknown | Promise, + options: { + // Default per-server connect + tools/list budget (a server's own + // connectTimeoutMs wins). Default 30_000; 0 disables. + connectTimeoutMs?: number + } = {}, ): Promise => { const clients = new Map() const tools: ToolSet = {} @@ -267,6 +305,7 @@ export const connectMcpServers = async ( ) } const description = t.description ?? '' + const readOnly = t.annotations?.readOnlyHint === true tools[prefixed] = dynamicTool({ description, inputSchema: jsonSchema(t.inputSchema as Parameters[0]), @@ -357,7 +396,10 @@ export const connectMcpServers = async ( }, ), }) - catalog.push({ name: prefixed, description, server: name }) + if (readOnly) { + markReadOnly(tools[prefixed]) + } + catalog.push({ name: prefixed, description, server: name, readOnly }) keys.push(prefixed) mounted++ } @@ -369,12 +411,76 @@ export const connectMcpServers = async ( } } + const listServerTools = (name: string, client: Client): Promise => + listAllTools(client, () => + log( + 'warn', + `[mcp] ${name}: tools/list still paginating after ${MAX_TOOL_LIST_PAGES} pages; the rest is ignored`, + ), + ) + const registerServerTools = async (name: string, client: Client): Promise => { - mountServerTools(name, client, (await client.listTools()).tools as IMcpToolDescriptor[]) + mountServerTools(name, client, await listServerTools(name, client)) } const openConnection = async (name: string, cfg: IMcpServerConfig): Promise => { + const configured = cfg.connectTimeoutMs ?? options.connectTimeoutMs + const timeoutMs = + typeof configured === 'number' && Number.isFinite(configured) && configured >= 0 + ? configured + : DEFAULT_MCP_CONNECT_TIMEOUT_MS + // `client` is shared with the watchdog below: a connect that times out + // closes whatever client it got as far as creating, and an attempt that + // completes AFTER the timeout closes its own client instead of mounting. + let client: Client | undefined + let timedOut = false + const attempt = connectOnce( + name, + cfg, + (c) => { + client = c + }, + () => timedOut, + ) + try { + return await (timeoutMs > 0 + ? withConnectTimeout(attempt, timeoutMs, () => { + timedOut = true + }) + : attempt) + } catch (err) { + // The client may be live even though we ended up here (listTools failing + // after a successful connect, or the watchdog firing). Close it, or the + // transport - an SSE stream or a spawned child process - outlives the + // failed connect. + if (client) { + await (client as Client).close().catch(() => {}) + } + const message = timedOut ? `connect timed out after ${timeoutMs}ms` : (err as Error).message + const needsAuthorization = !timedOut && err instanceof UnauthorizedError + log( + 'error', + needsAuthorization + ? `[mcp] ${name}: authorization required - complete the OAuth flow and reconnect` + : `[mcp] ${name}: failed to connect - ${message}`, + ) + return { name, error: message, needsAuthorization } + } + } + + const connectOnce = async ( + name: string, + cfg: IMcpServerConfig, + onClient: (client: Client) => void, + isTimedOut: () => boolean, + ): Promise => { let client: Client | undefined + const bail = async (): Promise => { + if (isTimedOut()) { + await client?.close().catch(() => {}) + throw new Error('connect timed out') + } + } try { let transport: Transport if (isStdioConfig(cfg)) { @@ -435,8 +541,11 @@ export const connectMcpServers = async ( transport.onerror = (err) => log('warn', `[mcp] ${name}: transport error - ${err.message}`) transport.onclose = () => log('warn', `[mcp] ${name}: transport closed`) + await bail() client = new Client({ name: clientName, version: CLIENT_VERSION }) + onClient(client) await client.connect(transport) + await bail() // Subscribe BEFORE the first list call: a server that mutates its tool // set during init would otherwise lose the notification in the small @@ -447,23 +556,14 @@ export const connectMcpServers = async ( }) } - return { name, client, listed: (await client.listTools()).tools as IMcpToolDescriptor[] } + const listed = await listServerTools(name, client) + await bail() + return { name, client, listed } } catch (err) { - // The client may be live even though we ended up here (listTools failing - // after a successful connect). Close it, or the transport - an SSE stream - // or a spawned child process - outlives the failed connect. if (client) { await client.close().catch(() => {}) } - const message = (err as Error).message - const needsAuthorization = err instanceof UnauthorizedError - log( - 'error', - needsAuthorization - ? `[mcp] ${name}: authorization required - complete the OAuth flow and reconnect` - : `[mcp] ${name}: failed to connect - ${message}`, - ) - return { name, error: message, needsAuthorization } + throw err } } @@ -506,6 +606,31 @@ export const connectMcpServers = async ( } } +// Race a connect attempt against a watchdog. The attempt keeps running in +// the background after a timeout (its own checks close the client), so its +// eventual rejection is swallowed here. +const withConnectTimeout = ( + attempt: Promise, + timeoutMs: number, + onTimeout: () => void, +): Promise => + new Promise((resolve, reject) => { + const timer = setTimeout(() => { + onTimeout() + reject(new Error(`connect timed out after ${timeoutMs}ms`)) + }, timeoutMs) + attempt.then( + (v) => { + clearTimeout(timer) + resolve(v) + }, + (err: unknown) => { + clearTimeout(timer) + reject(err) + }, + ) + }) + // Map common MCP mimeTypes to file extensions. Falls back to a generic // extension keyed by the part kind so the file at least carries a hint for // downstream tools. diff --git a/src/planner.ts b/src/planner.ts index 4cb9778..a5296cb 100644 --- a/src/planner.ts +++ b/src/planner.ts @@ -1,15 +1,17 @@ import { streamObject } from 'ai' import { z } from 'zod' -import type { IAgentInternalContext } from './internal.ts' +import { stageCallOptions } from './call-options.ts' +import type { EffectiveToolStrategy, IAgentInternalContext } from './internal.ts' import { - buildPlannerSystem, + DEFAULT_PLAN_STEP_CAP, buildPlannerUserPrompt, - withDomainContext, - type CatalogMode, + composePlannerSystem, + type PromptCatalogMode, } from './prompts.ts' +import { resolveToolStrategy } from './tool-search.ts' import { ATTR, withSpan } from './tracing.ts' import type { IConversationTurn, IPlan, IUsage } from './types.ts' -import { withRetry, withTimeout } from './utils.ts' +import { normalizeUsage, withRetry, withTimeout } from './utils.ts' export const PlanStepSchema = z.object({ id: z.string().min(1).describe('Short stable id, e.g. "s1", "fetch-companies"'), @@ -23,14 +25,60 @@ export const PlanStepSchema = z.object({ // NOTE: no .max() on steps — Anthropic's native structured output // (output_config.format.schema) rejects `maxItems` on array types. The cap -// is enforced via the planner prompt ("hard cap is 8") and a slice() below. -export const PLAN_STEP_HARD_CAP = 8 +// is enforced via the planner prompt ("hard cap is N") and a slice() below. +// Configurable per agent with `maxPlanSteps`. +export const PLAN_STEP_HARD_CAP = DEFAULT_PLAN_STEP_CAP export const PlanSchema = z.object({ thought: z.string().min(1).describe('One-paragraph reasoning about how to approach the request'), steps: z.array(PlanStepSchema).min(1), + // No .max() either (same Anthropic constraint). + skills: z + .array(z.string()) + .optional() + .describe('Names of skills from the SKILLS list that apply to this request'), }) +// The effective plan-step cap of an agent. +export const planStepCap = (ctx: IAgentInternalContext): number => { + const v = ctx.config.maxPlanSteps + return typeof v === 'number' && Number.isFinite(v) && v >= 1 ? Math.floor(v) : PLAN_STEP_HARD_CAP +} + +// The run's effective tool strategy (set by the runner), or resolved from +// the config for direct callers. +export const effectiveToolStrategy = (ctx: IAgentInternalContext): EffectiveToolStrategy => { + if (ctx.run) { + return ctx.run.strategy + } + const s = resolveToolStrategy(ctx.config, ctx.toolCatalog.length) + return s === 'search' && !ctx.findTools ? 'all' : s +} + +export const catalogModeFor = (strategy: EffectiveToolStrategy): PromptCatalogMode => + strategy === 'plan-narrowed' ? 'compact' : strategy === 'search' ? 'search' : 'full' + +// Keep only configured skill names (deduplicated); report the rest. +export const filterPlanSkills = ( + ctx: IAgentInternalContext, + names: string[] | undefined, +): string[] | undefined => { + if (!names?.length) { + return undefined + } + const known = new Set((ctx.skills ?? []).map((s) => s.name)) + const kept = [...new Set(names.filter((n) => known.has(n)))] + const dropped = names.filter((n) => !known.has(n)) + if (dropped.length) { + ctx.emit({ + type: 'log', + level: 'warn', + message: `[plan] dropped unknown skills: ${dropped.join(', ')}`, + }) + } + return kept.length ? kept : undefined +} + export const createInitialPlan = async ( input: string, history: IConversationTurn[] | undefined, @@ -39,8 +87,8 @@ export const createInitialPlan = async ( ): Promise => withSpan('agent.plan', { [ATTR.PHASE]: 'plan' }, async (span) => { const validNames = new Set(ctx.toolCatalog.map((t) => t.name)) - const catalogMode: CatalogMode = - ctx.config.toolSelectionStrategy === 'plan-narrowed' ? 'compact' : 'full' + const catalogMode = catalogModeFor(effectiveToolStrategy(ctx)) + const cap = planStepCap(ctx) const { object, usage } = await withRetry( () => streamPlanOnce(input, history, catalogMode, ctx, signal), @@ -55,17 +103,19 @@ export const createInitialPlan = async ( ctx.emit({ type: 'usage', phase: 'plan', usage }) span.setAttribute(ATTR.USAGE_TOTAL_TOKENS, usage.totalTokens) - const cappedSteps = object.steps.slice(0, PLAN_STEP_HARD_CAP) + const cappedSteps = object.steps.slice(0, cap) if (cappedSteps.length < object.steps.length) { ctx.emit({ type: 'log', level: 'warn', - message: `[plan] model returned ${object.steps.length} steps; truncated to hard cap ${PLAN_STEP_HARD_CAP}`, + message: `[plan] model returned ${object.steps.length} steps; truncated to hard cap ${cap}`, }) } + const skills = filterPlanSkills(ctx, object.skills) return { thought: object.thought, + ...(skills ? { skills } : {}), steps: cappedSteps.map((s) => { if (!s.suggestedTools?.length) { return s @@ -98,15 +148,26 @@ export const createInitialPlan = async ( const streamPlanOnce = async ( input: string, history: IConversationTurn[] | undefined, - catalogMode: CatalogMode, + catalogMode: PromptCatalogMode, ctx: IAgentInternalContext, signal?: AbortSignal, ): Promise<{ object: z.infer; usage: IUsage }> => { + const call = stageCallOptions( + ctx.config, + 'planner', + composePlannerSystem({ + mode: catalogMode, + catalog: ctx.toolCatalog, + maxSteps: planStepCap(ctx), + domain: ctx.config.systemPrompt, + skills: ctx.skills, + }), + ) const result = streamObject({ model: ctx.plannerModel, schema: PlanSchema, - system: withDomainContext(buildPlannerSystem(catalogMode), ctx.config.systemPrompt), - prompt: buildPlannerUserPrompt(input, ctx.toolCatalog, history, catalogMode), + ...call, + prompt: buildPlannerUserPrompt(input, history), abortSignal: withTimeout(signal, ctx.config.llmTimeoutMs ?? 0), }) @@ -168,12 +229,5 @@ const streamPlanOnce = async ( } } const [object, rawUsage] = await Promise.all([result.object, result.usage]) - return { - object, - usage: { - inputTokens: rawUsage.inputTokens ?? 0, - outputTokens: rawUsage.outputTokens ?? 0, - totalTokens: rawUsage.totalTokens ?? 0, - }, - } + return { object, usage: normalizeUsage(rawUsage) } } diff --git a/src/prompts.ts b/src/prompts.ts index c5d6ab9..77318a3 100644 --- a/src/prompts.ts +++ b/src/prompts.ts @@ -1,15 +1,47 @@ -import type { IConversationTurn, IPlan, IPlanStep, IStepResult } from './types.ts' +import { isSummaryTurn } from './compaction.ts' +import { renderActiveSkills, renderSkillsIndex } from './skills.ts' +import { FIND_TOOLS_TOOL, renderSearchCatalog } from './tool-search.ts' +import type { + IConversationTurn, + IPlan, + IPlanStep, + ISkill, + IStepResult, + IToolCatalogEntry, +} from './types.ts' + +// ── Layout ─────────────────────────────────────────────────────────────── +// Every stage splits its prompt in two: +// SYSTEM - run-stable content only: role instructions, domain context, +// skills index, tool catalogue (where the stage shows one) and the +// active skills. Identical across the calls of a stage within a +// run, so provider prefix caches (OpenAI / Gemini implicit, +// Anthropic via an explicit breakpoint) hit. +// USER - everything dynamic: history, request, plan, trace, state. const HISTORY_TURN_LIMIT = 8 const HISTORY_CONTENT_LIMIT = 1500 +// A compaction summary carries more than an ordinary turn; give it room. +const HISTORY_SUMMARY_LIMIT = 6000 export const renderHistory = (history: IConversationTurn[] | undefined): string => { if (!history?.length) { return '(no prior conversation)' } - const tail = history.slice(-HISTORY_TURN_LIMIT) - const skipped = history.length - tail.length + // A leading compaction summary is always kept (it stands for every turn + // before it), and the 8-turn window applies to the rest. + const summary = isSummaryTurn(history[0]) ? history[0] : undefined + const rest = summary ? history.slice(1) : history + const tail = rest.slice(-HISTORY_TURN_LIMIT) + const skipped = rest.length - tail.length const lines: string[] = [] + if (summary) { + const content = + summary.content.length > HISTORY_SUMMARY_LIMIT + ? `${summary.content.slice(0, HISTORY_SUMMARY_LIMIT)}... [truncated]` + : summary.content + lines.push(`${summary.role}: ${content}`) + } if (skipped > 0) { lines.push(`(${skipped} earlier turns omitted)`) } @@ -30,6 +62,10 @@ const CATALOG_BUDGETS = { export type CatalogMode = keyof typeof CATALOG_BUDGETS +// How a stage renders the catalogue: the two classic budgets, or the grouped +// short form of the 'search' strategy. +export type PromptCatalogMode = CatalogMode | 'search' + export const renderToolCatalog = ( catalog: { name: string; description: string }[], mode: CatalogMode = 'full', @@ -50,6 +86,9 @@ export const renderToolCatalog = ( return lines.join('\n') } +export const renderCatalogFor = (catalog: IToolCatalogEntry[], mode: PromptCatalogMode): string => + mode === 'search' ? renderSearchCatalog(catalog) : renderToolCatalog(catalog, mode) + const TOOL_OUTPUT_BUDGET = 1500 const truncateOutput = (output: unknown): string => { @@ -59,6 +98,9 @@ const truncateOutput = (output: unknown): string => { } catch { s = String(output) } + if (s === undefined) { + s = String(output) + } return s.length > TOOL_OUTPUT_BUDGET ? `${s.slice(0, TOOL_OUTPUT_BUDGET)}... [truncated, ${s.length - TOOL_OUTPUT_BUDGET} chars]` : s @@ -78,11 +120,30 @@ const renderTraceEntry = (r: IStepResult, idx: number): string => { ].join('\n') } -export const renderTrace = (trace: IStepResult[]): string => { +// Render steps whose first element sits at `offset` in the full trace. +export const renderTraceSteps = (steps: IStepResult[], offset = 0): string => + steps.map((r, i) => renderTraceEntry(r, offset + i)).join('\n') + +// A compacted view of the trace: steps [0, upTo) are represented by a +// running summary, the rest verbatim. +export interface ITraceView { + summary?: string + upTo: number +} + +export const renderTrace = (trace: IStepResult[], view?: ITraceView): string => { if (trace.length === 0) { return '(no steps executed yet)' } - return trace.map(renderTraceEntry).join('\n') + const from = view?.summary ? Math.min(Math.max(view.upTo, 0), trace.length) : 0 + if (from === 0) { + return renderTraceSteps(trace) + } + const lines = [`Summary of earlier steps (1-${from}): ${view!.summary}`] + if (from < trace.length) { + lines.push(renderTraceSteps(trace.slice(from), from)) + } + return lines.join('\n') } export const renderPlan = (plan: IPlan): string => @@ -103,51 +164,127 @@ export const renderPlan = (plan: IPlan): string => export const withDomainContext = (base: string, systemPrompt: string | undefined): string => systemPrompt ? `${base}\n\nDomain context:\n${systemPrompt}` : base -export const PLANNER_SYSTEM_BASE = `You are the Planner of a multi-step agent system. +export const DEFAULT_PLAN_STEP_CAP = 8 + +const plannerBase = ( + maxSteps: number, +): string => `You are the Planner of an autonomous multi-step agent system. -Your only job is to decompose the user's request into a short ordered list of concrete actionable steps that a tool-using Executor can perform one at a time. +Your only job is to decompose the user's request into a short ordered list of concrete, actionable steps that a tool-using Executor can carry out one at a time, on its own, without asking the user anything. Rules: -1. Produce 1-5 steps for typical requests (hard cap is 8). Prefer FEWER, larger steps over many micro-steps. If the task is trivial and needs no tools, output a single step "Answer the user directly". +1. Produce 1-5 steps for typical requests (hard cap is ${maxSteps}). Prefer FEWER, larger steps over many micro-steps. 2. Each step must be self-contained, action-oriented, and verifiable. State what should be done and what the expected outcome is. -3. If a step needs a tool, suggest tool name(s) ONLY from the provided available-tools list. NEVER fabricate tool names. -4. If the request asks for information that is likely retrievable via the available tools, plan to use them. If no tool fits, plan to answer from general knowledge. -5. The last step must produce the deliverable for the user (do not append a separate "summarize" step - the system synthesizes the final answer). -6. Output strict JSON matching the requested schema. No prose outside JSON.` +3. Plan tool use whenever the available tools can obtain, check or act on what the request needs. A question that needs data the tools can retrieve IS actionable: plan the lookups, never answer it from memory. Only when no tool is relevant (greetings, thanks, small talk, or a question fully answerable from the conversation or general knowledge) output a single step "Answer the user directly". +4. If a step needs a tool, suggest tool name(s) ONLY from the provided available-tools list. NEVER fabricate tool names. +5. Never plan a step that asks the user for clarification, confirmation or permission - the agent runs autonomously and the host handles tool consent. If the request is ambiguous, pick the most reasonable interpretation and state the assumption in the step description. +6. The last step must produce the deliverable for the user (do not append a separate "summarize" step - the system synthesizes the final answer). +7. Output strict JSON matching the requested schema. No prose outside JSON.` + +export const PLANNER_SYSTEM_BASE = plannerBase(DEFAULT_PLAN_STEP_CAP) const PLANNER_NARROWED_ADDENDUM = ` ADDITIONAL RULE (tool-narrowed mode): The agent narrows the active tool set per-step using suggestedTools. You MUST set suggestedTools to the EXACT tool names the executor will use for that step. If a step is reasoning-only and needs no tools, leave suggestedTools empty. Missing or wrong suggestedTools will leave the executor without the tools it needs.` -export const buildPlannerSystem = (mode: CatalogMode): string => - mode === 'compact' ? PLANNER_SYSTEM_BASE + PLANNER_NARROWED_ADDENDUM : PLANNER_SYSTEM_BASE +const PLANNER_SEARCH_ADDENDUM = ` -export const buildPlannerUserPrompt = ( - input: string, - toolCatalog: { name: string; description: string }[], - history?: IConversationTurn[], - catalogMode: CatalogMode = 'full', +ADDITIONAL RULE (tool-search mode): +The tool catalogue is large; the list below may be abridged. suggestedTools are optional hints: name tools you are sure fit the step, the executor loads them up front and can discover any other tool itself with ${FIND_TOOLS_TOOL}. Never invent names.` + +const PLANNER_SKILLS_ADDENDUM = ` + +SKILL SELECTION: +Skills are packaged instructions for specific kinds of work. Set "skills" to the names of the skills from the SKILLS list below that apply to this request (leave it empty when none does). Their instructions are given to the executor.` + +export const buildPlannerSystem = ( + mode: PromptCatalogMode, + maxSteps = DEFAULT_PLAN_STEP_CAP, ): string => + plannerBase(maxSteps) + + (mode === 'compact' + ? PLANNER_NARROWED_ADDENDUM + : mode === 'search' + ? PLANNER_SEARCH_ADDENDUM + : '') + +export interface ISystemParts { + // config.systemPrompt + domain?: string + skills?: ISkill[] + activeSkills?: string[] +} + +const skillsIndexSection = (skills: ISkill[] | undefined): string => + skills?.length ? `\n\nSKILLS:\n${renderSkillsIndex(skills)}` : '' + +const activeSkillsSection = ( + skills: ISkill[] | undefined, + active: string[] | undefined, +): string => { + if (!skills?.length || !active?.length) { + return '' + } + const text = renderActiveSkills(skills, active) + return text ? `\n\nACTIVE SKILLS (follow these instructions where they apply):\n\n${text}` : '' +} + +export const composePlannerSystem = ( + parts: ISystemParts & { + mode: PromptCatalogMode + catalog: IToolCatalogEntry[] + maxSteps?: number + }, +): string => + withDomainContext( + buildPlannerSystem(parts.mode, parts.maxSteps) + + (parts.skills?.length ? PLANNER_SKILLS_ADDENDUM : ''), + parts.domain, + ) + + skillsIndexSection(parts.skills) + + `\n\nAvailable tools:\n${renderCatalogFor(parts.catalog, parts.mode)}` + +export const buildPlannerUserPrompt = (input: string, history?: IConversationTurn[]): string => [ history?.length ? `Conversation history:\n${renderHistory(history)}\n` : '', `User request:\n${input}`, '', - `Available tools:\n${renderToolCatalog(toolCatalog, catalogMode)}`, - '', 'Produce the plan now.', ] .filter(Boolean) .join('\n') -export const EXECUTOR_SYSTEM = `You are the Executor of a multi-step agent system. You receive ONE step at a time and you must accomplish only that step. +export const EXECUTOR_SYSTEM = `You are the Executor of an autonomous multi-step agent system. You receive ONE step at a time and you must accomplish only that step. Rules: -1. Stay focused on the CURRENT step. Do not jump ahead, do not redo finished steps. -2. Use the available tools when they help. Call them with valid arguments. Read prior step results in the trace before re-fetching the same data. -3. When the step is complete, write a short concrete "step result" describing what you found / did. Include identifiers, names, or key data the next step might need. -4. If the step is impossible with the available tools, or if you are otherwise blocked, explain the blocker briefly and end your reply with the literal token [BLOCKER] on its own line. The system uses this token (language-independent) to invoke the Replanner. Do not fabricate data. -5. Be concise. Do not narrate your reasoning at length - the Replanner reads only your final summary.` +1. Stay focused on the CURRENT step. Do not jump ahead, do not redo finished steps. Read prior step results in the trace before re-fetching the same data. +2. Act on your own. Look things up with the available tools instead of guessing. Never ask the user questions or for confirmation - nobody answers mid-run, and tool consent is handled by the host: just call the tool you need. +3. When details are missing, choose sensible defaults and state the assumptions you made in the step result. +4. Call tools with valid arguments. If a call fails, fix the arguments or try another tool before giving up. If a tool call is denied, do not retry it; continue without it. +5. When the step is complete, write a short concrete "step result" describing what you found / did. Include identifiers, names, or key data the next step might need. Do not fabricate data. +6. Only when the step is truly impossible (missing credentials or permissions, a denied tool, data that does not exist or cannot be reached), explain the blocker briefly and end your reply with the literal token [BLOCKER] on its own line. The system uses this token (language-independent) to invoke the Replanner. +7. Be concise. Do not narrate your reasoning at length - the Replanner reads only your final summary.` + +const EXECUTOR_SEARCH_NOTE = ` + +TOOLS: +Only part of the tool catalogue is loaded. When you need a capability that is not among your tools, call ${FIND_TOOLS_TOOL} with a few keywords; the matching tools become callable from your next step on.` + +const EXECUTOR_SKILLS_NOTE = ` + +SKILLS: +Packaged instructions for specific kinds of work. When one of these applies to the current step and is not active yet, call load_skill with its name before doing the work; read_skill_file reads files it bundles.` + +export const composeExecutorSystem = ( + parts: ISystemParts & { searchMode?: boolean; toolCount?: number }, +): string => + withDomainContext(EXECUTOR_SYSTEM, parts.domain) + + (parts.skills?.length ? `${EXECUTOR_SKILLS_NOTE}\n${renderSkillsIndex(parts.skills)}` : '') + + (parts.searchMode + ? `${EXECUTOR_SEARCH_NOTE}${parts.toolCount ? ` The catalogue has ${parts.toolCount} tools.` : ''}` + : '') + + activeSkillsSection(parts.skills, parts.activeSkills) export const buildExecutorUserPrompt = ( input: string, @@ -155,6 +292,7 @@ export const buildExecutorUserPrompt = ( step: IPlanStep, trace: IStepResult[], history?: IConversationTurn[], + view?: ITraceView, ): string => [ history?.length ? `Conversation history:\n${renderHistory(history)}\n` : '', @@ -162,7 +300,7 @@ export const buildExecutorUserPrompt = ( '', `Overall plan:\n${renderPlan(plan)}`, '', - `Trace so far:\n${renderTrace(trace)}`, + `Trace so far:\n${renderTrace(trace, view)}`, '', `CURRENT STEP to execute (id=${step.id}): ${step.description}`, `Expected outcome: ${step.expectedOutcome}`, @@ -173,56 +311,69 @@ export const buildExecutorUserPrompt = ( .filter(Boolean) .join('\n') -export const REPLANNER_SYSTEM = `You are the Replanner of a multi-step agent system. After each Executor step you decide what should happen next. +export const REPLANNER_SYSTEM = `You are the Replanner of an autonomous multi-step agent system. After an Executor step you decide what should happen next. You have three options: - "continue": the next planned step is still appropriate. - "revise": the plan is wrong or incomplete given what we now know - produce a NEW plan covering only the REMAINING work (do not include already-completed steps). The new plan must follow the same rules as the original Planner. -- "finish": we already have enough information to answer the user. The system will then synthesize the final answer from the trace. +- "finish": we already have enough information to answer the user, or nothing more can be done. The system will then synthesize the final answer from the trace. Rules: 1. Prefer "continue" when the original plan still applies. Revise only when needed. 2. Prefer "finish" as soon as the user's request is satisfied - do not run unnecessary extra steps. -3. When revising, the new plan must NOT repeat already-completed work; it covers only what is still needed. -4. Output strict JSON matching the schema. No prose outside JSON.` +3. When a step failed or was blocked, prefer revising around the failure (another tool, other arguments, another source, a reasonable assumption) over finishing with nothing. Finish only when no workable alternative remains. +4. Never revise into a step that asks the user for input, clarification or permission - the agent runs autonomously. +5. When revising, the new plan must NOT repeat already-completed work; it covers only what is still needed. +6. Output strict JSON matching the schema. No prose outside JSON.` + +export const composeReplannerSystem = ( + parts: ISystemParts & { mode: PromptCatalogMode; catalog: IToolCatalogEntry[] }, +): string => + withDomainContext(REPLANNER_SYSTEM, parts.domain) + + `\n\nAvailable tools (for the revise option):\n${renderCatalogFor(parts.catalog, parts.mode)}` + + activeSkillsSection(parts.skills, parts.activeSkills) export const buildReplannerUserPrompt = ( input: string, plan: IPlan, trace: IStepResult[], nextStep: IPlanStep | null, - toolCatalog: { name: string; description: string }[], - catalogMode: CatalogMode = 'full', + view?: ITraceView, ): string => [ `Original user request:\n${input}`, '', `Current plan:\n${renderPlan(plan)}`, '', - `Completed steps:\n${renderTrace(trace)}`, + `Completed steps:\n${renderTrace(trace, view)}`, '', nextStep ? `Next planned step: [${nextStep.id}] ${nextStep.description}` : 'There is no next step in the current plan.', '', - `Available tools (for revise option):\n${renderToolCatalog(toolCatalog, catalogMode)}`, - '', 'Decide now: continue, revise, or finish.', ].join('\n') export const SYNTHESIZER_SYSTEM = `You are the Synthesizer. Produce the final answer for the user from the agent's plan and execution trace. Rules: -1. Answer the user directly and concisely. Do NOT mention "steps", "plans", or internal mechanics unless the user explicitly asked for them. -2. Use facts from the trace verbatim when accuracy matters (names, IDs, numbers, quotes). -3. If the trace shows the request could not be completed, say so plainly and explain what blocked it. -4. Use the user's language.` +1. Answer the user directly and concisely: report what was done and what was found. Do NOT mention "steps", "plans", or internal mechanics unless the user explicitly asked for them. +2. Use facts from the trace verbatim when accuracy matters (names, IDs, numbers, quotes). Never invent results the trace does not contain. +3. Mention the assumptions the agent made, briefly. +4. If the trace shows the request (or part of it) could not be completed, say so plainly and explain what blocked it. +5. Do not end with questions or offers unless the request genuinely cannot be completed without input from the user. +6. Use the user's language.` + +export const composeSynthesizerSystem = (parts: ISystemParts): string => + withDomainContext(SYNTHESIZER_SYSTEM, parts.domain) + + activeSkillsSection(parts.skills, parts.activeSkills) export const buildSynthesizerUserPrompt = ( input: string, plan: IPlan, trace: IStepResult[], history?: IConversationTurn[], + view?: ITraceView, ): string => [ history?.length ? `Conversation history:\n${renderHistory(history)}\n` : '', @@ -230,7 +381,7 @@ export const buildSynthesizerUserPrompt = ( '', `Plan that was executed:\n${renderPlan(plan)}`, '', - `Execution trace:\n${renderTrace(trace)}`, + `Execution trace:\n${renderTrace(trace, view)}`, '', 'Write the final answer for the user now.', ] diff --git a/src/replanner.ts b/src/replanner.ts index 7b7b4dc..3e68f72 100644 --- a/src/replanner.ts +++ b/src/replanner.ts @@ -1,11 +1,12 @@ import { generateObject } from 'ai' import { z } from 'zod' +import { stageCallOptions } from './call-options.ts' import type { IAgentInternalContext } from './internal.ts' -import { PlanSchema } from './planner.ts' -import { REPLANNER_SYSTEM, buildReplannerUserPrompt, withDomainContext } from './prompts.ts' +import { PlanSchema, catalogModeFor, effectiveToolStrategy } from './planner.ts' +import { buildReplannerUserPrompt, composeReplannerSystem } from './prompts.ts' import { ATTR, withSpan } from './tracing.ts' import type { IPlan, IPlanStep, IStepResult, IUsage } from './types.ts' -import { withRetry, withTimeout } from './utils.ts' +import { normalizeUsage, withRetry, withTimeout } from './utils.ts' export const DecisionSchema = z.discriminatedUnion('mode', [ z.object({ @@ -34,29 +35,32 @@ export const decideNextAction = async ( signal?: AbortSignal, ): Promise => withSpan('agent.replan', { [ATTR.PHASE]: 'replan' }, async (span) => { - const catalogMode = ctx.config.toolSelectionStrategy === 'plan-narrowed' ? 'compact' : 'full' + const catalogMode = catalogModeFor(effectiveToolStrategy(ctx)) + const view = ctx.run + ? { summary: ctx.run.traceSummary, upTo: ctx.run.traceSummaryUpTo } + : undefined const { object, usage } = await withRetry( async () => { + const call = stageCallOptions( + ctx.config, + 'replanner', + composeReplannerSystem({ + mode: catalogMode, + catalog: ctx.toolCatalog, + domain: ctx.config.systemPrompt, + skills: ctx.skills, + activeSkills: ctx.run?.activeSkills, + }), + ) const r = await generateObject({ model: ctx.plannerModel, schema: DecisionSchema, - system: withDomainContext(REPLANNER_SYSTEM, ctx.config.systemPrompt), - prompt: buildReplannerUserPrompt( - input, - plan, - trace, - nextStep, - ctx.toolCatalog, - catalogMode, - ), + ...call, + prompt: buildReplannerUserPrompt(input, plan, trace, nextStep, view), abortSignal: withTimeout(signal, ctx.config.llmTimeoutMs ?? 0), }) - const usage: IUsage = { - inputTokens: r.usage.inputTokens ?? 0, - outputTokens: r.usage.outputTokens ?? 0, - totalTokens: r.usage.totalTokens ?? 0, - } + const usage: IUsage = normalizeUsage(r.usage) return { object: r.object, usage } }, { diff --git a/src/runner.ts b/src/runner.ts index d4e0f46..04e77e7 100644 --- a/src/runner.ts +++ b/src/runner.ts @@ -2,25 +2,32 @@ import { randomUUID } from 'node:crypto' import { rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import path from 'node:path' +import { compactionMaxTokens } from './call-options.ts' +import { compactHistory, estimateTokens, resolveCompaction, summarizeTrace } from './compaction.ts' import { runContext } from './context.ts' import { executeStep } from './executor.ts' -import { createInitialPlan } from './planner.ts' +import { checkLimits, resolveLimits } from './limits.ts' +import { createInitialPlan, filterPlanSkills, planStepCap } from './planner.ts' +import { renderTrace, renderTraceSteps } from './prompts.ts' import { decideNextAction } from './replanner.ts' +import { activateSkill } from './skills.ts' import { synthesizeAnswer } from './synthesizer.ts' -import type { IAgentInternalContext } from './internal.ts' +import { resolveToolStrategy } from './tool-search.ts' +import type { EffectiveToolStrategy, IAgentInternalContext, IRunState } from './internal.ts' import type { AgentEvent, + BudgetKind, IAgentRunOptions, IAgentRunResult, + IConversationTurn, IPersistence, IPlan, IRunSnapshot, IStepResult, - IUsage, ReplanTrigger, } from './types.ts' import { ATTR, withSpan } from './tracing.ts' -import { combineSignals } from './utils.ts' +import { accumulateUsage, combineSignals, normalizeUsage } from './utils.ts' // The default ('failure') replan trigger fires when: // - the executor explicitly signalled a blocker (via the [BLOCKER] sentinel, @@ -225,6 +232,13 @@ export const runAgentLoop = async ( }) } +// The run's effective tool strategy: 'auto' resolved against the live +// catalogue size, and 'search' only when find_tools is available. +const runToolStrategy = (ctx: IAgentInternalContext): EffectiveToolStrategy => { + const s = resolveToolStrategy(ctx.config, ctx.toolCatalog.length) + return s === 'search' && !ctx.findTools ? 'all' : s +} + const runAgentLoopInner = async ( ctx: IAgentInternalContext, options: IAgentRunOptions, @@ -237,29 +251,43 @@ const runAgentLoopInner = async ( // History follows the same rule for symmetry - the original run's history // is what shaped the saved trace, so we keep it. Documented in README. const input = resumed?.input ?? options.input - const history = resumed?.history + let history: IConversationTurn[] | undefined = resumed?.history ? [...resumed.history] : options.history ? [...options.history] : undefined const onEvent = options.onEvent ?? (() => {}) + const compaction = resolveCompaction(ctx.config.compaction) + const limits = resolveLimits(ctx.config) + const maxToolCalls = ctx.config.maxToolCalls - const totalUsage: IUsage = resumed - ? { ...resumed.usage } - : { inputTokens: 0, outputTokens: 0, totalTokens: 0 } + // Every IUsage detail is filled (normalizeUsage) so accumulation never + // produces NaN on a snapshot written by an older version. + const totalUsage = normalizeUsage(resumed?.usage) const startedAt = runContext.getStore()?.startedAt ?? Date.now() const runId = runContext.getStore()?.runId ?? '' + // Per-run state shared with the stages and (through the run context) with + // the tool wrappers. + const state: IRunState = { + usage: totalUsage, + toolCalls: resumed?.toolCallCount ?? 0, + strategy: runToolStrategy(ctx), + discovered: [], + activeSkills: [...(resumed?.activeSkills ?? [])], + traceSummary: resumed?.traceSummary, + traceSummaryUpTo: resumed?.traceSummaryUpTo ?? 0, + } + // proxiedCtx must be in scope BEFORE callPersistence so the persistence // error log carries the runId tag (the proxy attaches it; ctx.emit alone // does not). Order of declaration matters here. const proxiedCtx: IAgentInternalContext = { ...ctx, + run: state, emit: (event: AgentEvent) => { if (event.type === 'usage') { - totalUsage.inputTokens += event.usage.inputTokens - totalUsage.outputTokens += event.usage.outputTokens - totalUsage.totalTokens += event.usage.totalTokens + accumulateUsage(totalUsage, event.usage) } const tagged = { ...event, runId: runContext.getStore()?.runId } try { @@ -270,6 +298,14 @@ const runAgentLoopInner = async ( } catch {} }, } + const store = runContext.getStore() + if (store) { + store.emit = proxiedCtx.emit + store.state = state + } + const activate = (name: string, by: 'plan' | 'tool'): void => { + activateSkill(ctx.skills ?? [], name, by, state, proxiedCtx.emit) + } // Persistence facade. Hooks may be async; we await so a slow store // back-pressures the run. Write-hook failures are logged but never @@ -294,6 +330,101 @@ const runAgentLoopInner = async ( } } + // Automatic history compaction (fresh runs only - a resumed run already + // carries the history its trace was built from). The run works on the + // compacted copy; the caller's array is never mutated and gets the copy + // back as result.compactedHistory. + let compactedHistory: IConversationTurn[] | undefined + if (!resumed && history?.length && compaction.auto) { + const r = await compactHistory(history, ctx.synthesizerModel, { + thresholdTokens: compaction.thresholdTokens, + keepRecentTurns: compaction.keepRecentTurns, + summaryMaxTokens: compactionMaxTokens(ctx.config, compaction.summaryMaxTokens), + signal, + timeoutMs: ctx.config.llmTimeoutMs, + onUsage: (usage) => proxiedCtx.emit({ type: 'usage', phase: 'compact', usage }), + }) + if (r.compacted) { + history = r.history + compactedHistory = r.history + proxiedCtx.emit({ + type: 'context.compacted', + scope: 'history', + beforeTokens: r.beforeTokens, + afterTokens: r.afterTokens, + }) + } + } + + // Automatic trace compaction, before each executor / replanner / + // synthesizer call: once the rendered trace crosses the threshold, every + // step but the last keepRecentSteps is folded into a running summary. + // The trace itself stays intact (result, persistence, events). + const maybeCompactTrace = async (trace: IStepResult[]): Promise => { + if (!compaction.auto || signal?.aborted) { + return + } + const view = () => ({ summary: state.traceSummary, upTo: state.traceSummaryUpTo }) + const beforeTokens = estimateTokens(renderTrace(trace, view())) + if (beforeTokens <= compaction.thresholdTokens) { + return + } + const upTo = trace.length - compaction.keepRecentSteps + const from = state.traceSummary ? state.traceSummaryUpTo : 0 + if (upTo <= from) { + return + } + const r = await summarizeTrace( + trace.slice(from, upTo), + state.traceSummary, + ctx.synthesizerModel, + { + summaryMaxTokens: compactionMaxTokens(ctx.config, compaction.summaryMaxTokens), + signal, + timeoutMs: ctx.config.llmTimeoutMs, + render: renderTraceSteps, + offset: from, + }, + ) + if (!r) { + return + } + proxiedCtx.emit({ type: 'usage', phase: 'compact', usage: r.usage }) + state.traceSummary = r.summary + state.traceSummaryUpTo = upTo + proxiedCtx.emit({ + type: 'context.compacted', + scope: 'trace', + beforeTokens, + afterTokens: estimateTokens(renderTrace(trace, view())), + }) + } + + // Run-level budgets: token limits (legacy maxTotalTokens included) and the + // tool-call cap. Reported once per run. + let budgetReported = false + const budgetBreach = (): { kind: BudgetKind; tokens: number; cap: number } | undefined => { + const tokens = checkLimits(totalUsage, limits) + if (tokens) { + return tokens + } + if (typeof maxToolCalls === 'number' && maxToolCalls > 0 && state.toolCalls >= maxToolCalls) { + return { kind: 'tool-calls', tokens: state.toolCalls, cap: maxToolCalls } + } + return undefined + } + const reportBudget = (): boolean => { + const breach = budgetBreach() + if (!breach) { + return false + } + if (!budgetReported) { + budgetReported = true + proxiedCtx.emit({ type: 'budget.exceeded', ...breach }) + } + return true + } + let plan: IPlan if (resumed) { // Resume path: planner already ran, plan + trace are durable. The @@ -348,6 +479,9 @@ const runAgentLoopInner = async ( } } proxiedCtx.emit({ type: 'plan.created', plan }) + for (const name of plan.skills ?? []) { + activate(name, 'plan') + } } const trace: IStepResult[] = resumed ? [...resumed.trace] : [] @@ -360,7 +494,6 @@ const runAgentLoopInner = async ( let iterations = resumed?.iterations ?? 0 let revisions = resumed?.revisions ?? 0 const maxRevisions = ctx.config.maxRevisions ?? 2 - const tokenCap = ctx.config.maxTotalTokens // Build a complete snapshot for the current loop state. Centralises the // 14-field literal that used to be repeated at every persistence call site. @@ -379,6 +512,11 @@ const runAgentLoopInner = async ( stepIndex, iterations, revisions, + ...(state.traceSummary + ? { traceSummary: state.traceSummary, traceSummaryUpTo: state.traceSummaryUpTo } + : {}), + ...(state.activeSkills.length ? { activeSkills: [...state.activeSkills] } : {}), + ...(state.toolCalls ? { toolCallCount: state.toolCalls } : {}), ...extra, }) @@ -403,8 +541,7 @@ const runAgentLoopInner = async ( if (stepIndex >= currentPlan.steps.length) { break } - if (tokenCap && totalUsage.totalTokens >= tokenCap) { - proxiedCtx.emit({ type: 'budget.exceeded', tokens: totalUsage.totalTokens, cap: tokenCap }) + if (reportBudget()) { break } // Increment AFTER the early-break checks so iterations counts only @@ -413,6 +550,9 @@ const runAgentLoopInner = async ( iterations++ const step = currentPlan.steps[stepIndex] + if (store) { + store.currentStep = step + } proxiedCtx.emit({ type: 'step.start', step, index: stepIndex }) // Per-step abort: separate from the run-level signal so the user can @@ -436,6 +576,7 @@ const runAgentLoopInner = async ( let result: IStepResult try { + await maybeCompactTrace(trace) result = await executeStep(input, currentPlan, step, trace, history, proxiedCtx, stepSignal) } catch (err) { // Distinguish run-level abort (propagate) from step-level abort @@ -483,6 +624,13 @@ const runAgentLoopInner = async ( // serialized stepIndex / currentPlan reflect a stable next-loop-entry // state. See the three persistCheckpoint() calls below. + // A budget crossed during the step ends execution right here: no + // replanner call, straight to synthesis. + if (reportBudget()) { + stepIndex++ + break + } + const nextStep = currentPlan.steps[stepIndex + 1] ?? null const isLastPlannedStep = nextStep === null @@ -523,6 +671,7 @@ const runAgentLoopInner = async ( let decision try { + await maybeCompactTrace(trace) decision = await decideNextAction(input, currentPlan, trace, nextStep, proxiedCtx, signal) } catch (err) { proxiedCtx.emit({ type: 'error', error: asError(err), phase: 'replan' }) @@ -551,8 +700,11 @@ const runAgentLoopInner = async ( break } revisions++ - currentPlan = decision.newPlan + currentPlan = reviseInto(proxiedCtx, decision.newPlan) proxiedCtx.emit({ type: 'plan.revised', plan: currentPlan, reason: decision.reason }) + for (const name of currentPlan.skills ?? []) { + activate(name, 'plan') + } stepIndex = 0 await persistCheckpoint() continue @@ -563,6 +715,7 @@ const runAgentLoopInner = async ( let text: string try { + await maybeCompactTrace(trace) text = await synthesizeAnswer(input, currentPlan, trace, history, proxiedCtx, signal) } catch (err) { proxiedCtx.emit({ type: 'error', error: asError(err), phase: 'synthesize' }) @@ -584,7 +737,30 @@ const runAgentLoopInner = async ( }), ) - return { text, plan: currentPlan, trace, iterations, usage: totalUsage } + return { + text, + plan: currentPlan, + trace, + iterations, + usage: totalUsage, + ...(compactedHistory ? { compactedHistory } : {}), + } +} + +// A revised plan follows the planner's rules too: the same step cap and +// only configured skills. +const reviseInto = (ctx: IAgentInternalContext, plan: IPlan): IPlan => { + const cap = planStepCap(ctx) + if (plan.steps.length > cap) { + ctx.emit({ + type: 'log', + level: 'warn', + message: `[replan] revised plan has ${plan.steps.length} steps; truncated to hard cap ${cap}`, + }) + } + const skills = filterPlanSkills(ctx, plan.skills) + const { skills: _drop, ...rest } = plan + return { ...rest, steps: plan.steps.slice(0, cap), ...(skills ? { skills } : {}) } } const asError = (err: unknown): Error => diff --git a/src/skills.ts b/src/skills.ts new file mode 100644 index 0000000..0b496ce --- /dev/null +++ b/src/skills.ts @@ -0,0 +1,394 @@ +import { tool, type ToolSet } from 'ai' +import { readdir, readFile, stat } from 'node:fs/promises' +import path from 'node:path' +import { z } from 'zod' +import { markReadOnly } from './approval.ts' +import { runContext } from './context.ts' +import type { ISkill } from './types.ts' + +export const SKILL_NAME_PATTERN = /^[a-z0-9-]{1,64}$/ + +// Built-in tool names (reserved when skills are configured). +export const LOAD_SKILL_TOOL = 'load_skill' +export const READ_SKILL_FILE_TOOL = 'read_skill_file' + +// Budget for active skill instructions injected into system prompts. +export const ACTIVE_SKILLS_BUDGET_CHARS = 24_000 + +const normalizeSkillPath = (p: string): string => { + const posix = p.replace(/\\/g, '/').replace(/^\.\/+/, '') + const normalized = path.posix.normalize(posix) + if ( + !normalized || + normalized === '.' || + normalized.startsWith('/') || + normalized === '..' || + normalized.startsWith('../') + ) { + throw new Error(`skill file path must be relative and stay inside the skill: "${p}"`) + } + return normalized +} + +/** + * Validate a skill (name, description, content, files) and return a + * normalised copy. Throws a descriptive error on anything invalid. + */ +export const defineSkill = (skill: ISkill): ISkill => { + if (!skill || typeof skill !== 'object') { + throw new Error('skill must be an object { name, description, content, files? }') + } + const { name, description, content, files } = skill + if (typeof name !== 'string' || !SKILL_NAME_PATTERN.test(name)) { + throw new Error(`skill name must match ${SKILL_NAME_PATTERN} (got ${JSON.stringify(name)})`) + } + if (typeof description !== 'string' || !description.trim()) { + throw new Error(`skill "${name}": description is required`) + } + if (typeof content !== 'string') { + throw new Error(`skill "${name}": content must be a string`) + } + if (files !== undefined && !Array.isArray(files)) { + throw new Error(`skill "${name}": files must be an array of { path, content }`) + } + const seen = new Set() + const normalizedFiles = (files ?? []).map((f) => { + if (!f || typeof f.path !== 'string' || typeof f.content !== 'string') { + throw new Error(`skill "${name}": every file needs a string path and content`) + } + let p: string + try { + p = normalizeSkillPath(f.path) + } catch (err) { + throw new Error(`skill "${name}": ${(err as Error).message}`) + } + if (seen.has(p)) { + throw new Error(`skill "${name}": duplicate file "${p}"`) + } + seen.add(p) + return { path: p, content: f.content } + }) + return { + name, + description: description.trim(), + content, + ...(normalizedFiles.length ? { files: normalizedFiles } : {}), + } +} + +const unquote = (raw: string): string => { + const v = raw.trim() + if (v.length >= 2 && v.startsWith('"') && v.endsWith('"')) { + try { + return JSON.parse(v) as string + } catch { + return v.slice(1, -1) + } + } + if (v.length >= 2 && v.startsWith("'") && v.endsWith("'")) { + return v.slice(1, -1).replace(/''/g, "'") + } + // Plain scalar: strip a trailing " # comment". + return v.replace(/\s+#.*$/, '') +} + +// Minimal YAML subset: top-level `key: value` pairs with plain / quoted +// scalars and `>` / `|` block scalars (with optional -/+ chomping). +// Nested maps and lists are skipped. +const parseFrontmatter = (block: string): Record => { + const out: Record = {} + const lines = block.split('\n') + for (let i = 0; i < lines.length; i++) { + const line = lines[i] + const m = /^([A-Za-z0-9_-]+)\s*:(.*)$/.exec(line) + if (!m) { + continue + } + const key = m[1] + const rest = m[2].trim() + const block = /^([>|])([+-]?)\s*(#.*)?$/.exec(rest) + if (block) { + const collected: string[] = [] + while (i + 1 < lines.length && (/^\s/.test(lines[i + 1]) || lines[i + 1].trim() === '')) { + collected.push(lines[++i]) + } + while (collected.length && collected[collected.length - 1].trim() === '') { + collected.pop() + } + const indent = Math.min( + ...collected.filter((l) => l.trim()).map((l) => /^\s*/.exec(l)![0].length), + ) + const body = collected.map((l) => l.slice(Number.isFinite(indent) ? indent : 0)) + out[key] = + block[1] === '|' + ? body.join('\n') + : body + .join('\n') + .split(/\n\s*\n/) + .map((para) => para.replace(/\n/g, ' ').trim()) + .join('\n') + continue + } + out[key] = unquote(rest) + } + return out +} + +/** + * Parse a SKILL.md: YAML-ish frontmatter between `---` lines with `name:` + * and `description:`, the body is the content. Throws a clear error when the + * frontmatter, the name or the description is missing or the name invalid. + */ +export const parseSkillMarkdown = ( + markdown: string, + files?: { path: string; content: string }[], +): ISkill => { + const text = markdown.replace(/^\uFEFF/, '').replace(/\r\n?/g, '\n') + const m = /^\s*---[ \t]*\n([\s\S]*?)\n---[ \t]*(?:\n|$)/.exec(text) + if (!m) { + throw new Error('SKILL.md must start with a frontmatter block between "---" lines') + } + const meta = parseFrontmatter(m[1]) + if (!meta.name) { + throw new Error('SKILL.md frontmatter is missing "name"') + } + if (!meta.description) { + throw new Error(`SKILL.md "${meta.name}": frontmatter is missing "description"`) + } + return defineSkill({ + name: meta.name, + description: meta.description, + content: text.slice(m[0].length).trim(), + ...(files?.length ? { files } : {}), + }) +} + +const TEXT_EXTENSIONS: ReadonlySet = new Set([ + '.md', + '.txt', + '.json', + '.yaml', + '.yml', + '.csv', + '.ts', + '.js', + '.py', + '.sh', + '.html', + '.css', + '.xml', + '.toml', +]) +const MAX_SKILL_FILE_BYTES = 256 * 1024 + +const collectFiles = async ( + root: string, + dir: string, + out: { path: string; content: string }[], +): Promise => { + const entries = await readdir(dir, { withFileTypes: true }) + entries.sort((a, b) => a.name.localeCompare(b.name)) + for (const e of entries) { + if (e.name.startsWith('.') || e.name === 'node_modules') { + continue + } + const full = path.join(dir, e.name) + if (e.isDirectory()) { + await collectFiles(root, full, out) + continue + } + if (!e.isFile()) { + continue + } + const rel = path.relative(root, full).split(path.sep).join('/') + if (rel === 'SKILL.md' || !TEXT_EXTENSIONS.has(path.extname(e.name).toLowerCase())) { + continue + } + if ((await stat(full)).size > MAX_SKILL_FILE_BYTES) { + continue + } + out.push({ path: rel, content: await readFile(full, 'utf8') }) + } +} + +/** + * Load every `//SKILL.md`. Bundled files are the other text + * files under that skill folder (recursive; text extensions only; files over + * 256 KB skipped). Folders without a SKILL.md are ignored; duplicate skill + * names throw. + */ +export const loadSkillsFromDir = async (dir: string): Promise => { + const entries = await readdir(dir, { withFileTypes: true }) + entries.sort((a, b) => a.name.localeCompare(b.name)) + const skills: ISkill[] = [] + const names = new Set() + for (const e of entries) { + if (!e.isDirectory() || e.name.startsWith('.') || e.name === 'node_modules') { + continue + } + const skillDir = path.join(dir, e.name) + let markdown: string + try { + markdown = await readFile(path.join(skillDir, 'SKILL.md'), 'utf8') + } catch { + continue + } + const files: { path: string; content: string }[] = [] + await collectFiles(skillDir, skillDir, files) + let skill: ISkill + try { + skill = parseSkillMarkdown(markdown, files) + } catch (err) { + throw new Error(`${path.join(skillDir, 'SKILL.md')}: ${(err as Error).message}`) + } + if (names.has(skill.name)) { + throw new Error(`duplicate skill name "${skill.name}" in ${dir}`) + } + names.add(skill.name) + skills.push(skill) + } + return skills +} + +// Validate a configured skill list (shapes + unique names). +export const validateSkills = (skills: ISkill[] | undefined): ISkill[] => { + const out: ISkill[] = [] + const names = new Set() + for (const s of skills ?? []) { + const skill = defineSkill(s) + if (names.has(skill.name)) { + throw new Error(`duplicate skill name "${skill.name}"`) + } + names.add(skill.name) + out.push(skill) + } + return out +} + +const oneLine = (s: string, max: number): string => { + const flat = s.replace(/\s+/g, ' ').trim() + return flat.length > max ? `${flat.slice(0, max)}…` : flat +} + +// "- name: description" lines for the planner / executor system prompts. +export const renderSkillsIndex = (skills: ISkill[]): string => + skills.map((s) => `- ${s.name}: ${oneLine(s.description, 300)}`).join('\n') + +/** + * The instructions of the active skills, in activation order, within a char + * budget (the rest is clipped with a marker). + */ +export const renderActiveSkills = ( + skills: ISkill[], + active: string[], + budget = ACTIVE_SKILLS_BUDGET_CHARS, +): string => { + const byName = new Map(skills.map((s) => [s.name, s])) + const parts: string[] = [] + let used = 0 + for (const name of active) { + const skill = byName.get(name) + if (!skill) { + continue + } + const block = `### Skill: ${skill.name}\n${skill.content.trim()}` + const left = budget - used + if (left <= 0) { + parts.push(`### Skill: ${skill.name}\n… [skill instructions clipped: budget exhausted]`) + continue + } + if (block.length > left) { + parts.push(`${block.slice(0, left)}\n… [skill instructions clipped]`) + used = budget + continue + } + parts.push(block) + used += block.length + } + return parts.join('\n\n') +} + +/** + * Activate a skill for the rest of the current run (idempotent). Emits + * skill.activated once per skill and run. Returns false for an unknown name. + */ +export const activateSkill = ( + skills: ISkill[], + name: string, + by: 'plan' | 'tool', + target?: { activeSkills: string[] }, + emit?: (event: { type: 'skill.activated'; name: string; by: 'plan' | 'tool' }) => void, +): boolean => { + if (!skills.some((s) => s.name === name)) { + return false + } + if (target && !target.activeSkills.includes(name)) { + target.activeSkills.push(name) + emit?.({ type: 'skill.activated', name, by }) + } + return true +} + +// The built-in load_skill / read_skill_file tools. Read-only, never need +// approval; they activate skills in the CURRENT run via the run context. +export const createSkillTools = (skills: ISkill[]): ToolSet => { + if (!skills.length) { + return {} + } + const names = skills.map((s) => s.name) + const find = (name: string): ISkill => { + const skill = skills.find((s) => s.name === name) + if (!skill) { + throw new Error(`Unknown skill "${name}". Available skills: ${names.join(', ')}`) + } + return skill + } + return { + [LOAD_SKILL_TOOL]: markReadOnly( + tool({ + description: + 'Load the full instructions of a skill from the SKILLS list and activate it for the rest of the task. Call it before doing work a skill covers.', + inputSchema: z.object({ + name: z.string().describe(`Skill name, one of: ${names.join(', ')}`), + }), + execute: async ({ name }) => { + const skill = find(name) + const store = runContext.getStore() + activateSkill(skills, skill.name, 'tool', store?.state, store?.emit) + return { + name: skill.name, + content: skill.content, + files: (skill.files ?? []).map((f) => f.path), + } + }, + }), + ), + [READ_SKILL_FILE_TOOL]: markReadOnly( + tool({ + description: + 'Read a file bundled with a skill (paths are listed by load_skill). Returns the file content.', + inputSchema: z.object({ + name: z.string().describe('Skill name'), + path: z.string().describe('Skill-relative file path, as listed by load_skill'), + }), + execute: async ({ name, path: filePath }) => { + const skill = find(name) + let wanted: string + try { + wanted = normalizeSkillPath(filePath) + } catch (err) { + throw new Error((err as Error).message) + } + const file = (skill.files ?? []).find((f) => f.path === wanted) + if (!file) { + const listed = (skill.files ?? []).map((f) => f.path) + throw new Error( + `Skill "${skill.name}" has no file "${filePath}". Files: ${listed.length ? listed.join(', ') : '(none)'}`, + ) + } + return file.content + }, + }), + ), + } +} diff --git a/src/subagent-worker.ts b/src/subagent-worker.ts new file mode 100644 index 0000000..6a169f0 --- /dev/null +++ b/src/subagent-worker.ts @@ -0,0 +1,100 @@ +// Worker-thread entry of a subagent (createSubagentTool, isolation 'worker'). +// Bundled to dist/subagent-worker.js; run from source as +// src/subagent-worker.ts (the worker inherits --experimental-strip-types). +// +// Protocol: see ParentToWorkerMessage / WorkerToParentMessage in subagent.ts. +import { dynamicTool, jsonSchema, type ToolSet } from 'ai' +import { randomUUID } from 'node:crypto' +import { isMainThread, parentPort, type MessagePort } from 'node:worker_threads' +import { createAgent, type IAgent } from './agent.ts' +import { markReadOnly } from './approval.ts' +import { toCloneable, type ParentToWorkerMessage, type WorkerToParentMessage } from './subagent.ts' +import { errorMessage } from './utils.ts' + +export const serveSubagentWorker = (port: MessagePort): void => { + const ac = new AbortController() + const pending = new Map void>() + let started = false + const post = (msg: WorkerToParentMessage): void => { + try { + port.postMessage(msg) + } catch { + port.postMessage(toCloneable(msg)) + } + } + + const run = async (msg: Extract): Promise => { + // Proxied host tools: the child sees the schema, the call runs on the + // parent thread and the result comes back as a tool-result message. + const tools: ToolSet = {} + for (const p of msg.proxied) { + tools[p.name] = markReadOnly( + dynamicTool({ + description: p.description, + inputSchema: jsonSchema(p.inputSchema as Parameters[0]), + execute: (input, options) => + new Promise((resolve, reject) => { + const callId = randomUUID() + const signal = options?.abortSignal + const onAbort = (): void => { + pending.delete(callId) + reject(signal?.reason ?? new DOMException('Aborted', 'AbortError')) + } + if (signal?.aborted) { + onAbort() + return + } + signal?.addEventListener('abort', onAbort, { once: true }) + pending.set(callId, (r) => { + pending.delete(callId) + signal?.removeEventListener('abort', onAbort) + if (r.ok) { + resolve(r.output) + } else { + reject( + new Error(typeof r.output === 'string' ? r.output : JSON.stringify(r.output)), + ) + } + }) + post({ type: 'tool-call', callId, name: p.name, input: toCloneable(input) }) + }), + }), + p.readOnly === true, + ) + } + let agent: IAgent | undefined + try { + agent = await createAgent( + { ...msg.config, ...(msg.proxied.length ? { tools } : {}) }, + (event) => post({ type: 'event', event: toCloneable(event) }), + ) + const r = await agent.run({ input: msg.task, signal: ac.signal }) + post({ type: 'done', text: r.text, usage: toCloneable(r.usage), iterations: r.iterations }) + } catch (err) { + post({ type: 'error', error: errorMessage(err) }) + } finally { + await agent?.close({ timeoutMs: 2_000 }).catch(() => {}) + } + } + + port.on('message', (msg: ParentToWorkerMessage) => { + switch (msg.type) { + case 'run': + if (!started) { + started = true + void run(msg) + } + break + case 'abort': + ac.abort() + break + case 'tool-result': + pending.get(msg.callId)?.({ ok: msg.ok, output: msg.output }) + break + } + }) +} + +if (!isMainThread && parentPort) { + serveSubagentWorker(parentPort) +} diff --git a/src/subagent.ts b/src/subagent.ts new file mode 100644 index 0000000..dbcdd75 --- /dev/null +++ b/src/subagent.ts @@ -0,0 +1,440 @@ +import { asSchema, tool, type Tool, type ToolExecutionOptions, type ToolSet } from 'ai' +import { randomUUID } from 'node:crypto' +import { AsyncResource } from 'node:async_hooks' +import { Worker } from 'node:worker_threads' +import { z } from 'zod' +import { createAgent } from './agent.ts' +import { isReadOnlyTool, markReadOnly } from './approval.ts' +import { runContext } from './context.ts' +import type { AgentEvent, EventHandler, IAgentConfig, IUsage } from './types.ts' +import { clipText, emptyUsage, errorMessage, normalizeUsage } from './utils.ts' + +export type SubagentIsolation = 'worker' | 'in-process' + +export interface ISubagentToolOptions { + // Tool name the parent model calls. + name: string + // When the parent should delegate to this subagent. + description: string + // The child agent's config. With 'worker' isolation it must be + // structured-cloneable: no functions anywhere (getHeaders, authProvider, + // fetch, sanitizers, persistence, onRequest, native tools...). + config: IAgentConfig + // 'worker' (default): the child runs in its own worker_thread - its CPU + // work and crashes stay off the parent's event loop. 'in-process': the + // child is a plain agent on this thread. + isolation?: SubagentIsolation + // Host tools for the child. In-process: merged into config.tools. Worker: + // PROXIED - the child sees their schemas, every call runs here on the + // parent thread. + tools?: ToolSet + // Parallel calls of this tool beyond this many queue. Default 4. + maxConcurrent?: number + // Abort the child after this many ms. Default: none. + timeoutMs?: number + // The child's final text is clipped to this. Default 8_000. + outputMaxChars?: number + // Mark the tool read-only for tool approval. Default false. + readOnly?: boolean +} + +// ── worker protocol (JSON-cloneable) ───────────────────────────────────── +export interface IProxiedToolDescriptor { + name: string + description: string + inputSchema: unknown + readOnly?: boolean +} + +export type ParentToWorkerMessage = + | { type: 'run'; task: string; config: IAgentConfig; proxied: IProxiedToolDescriptor[] } + | { type: 'abort' } + | { type: 'tool-result'; callId: string; ok: boolean; output: unknown } + +export type WorkerToParentMessage = + | { type: 'event'; event: AgentEvent } + | { type: 'tool-call'; callId: string; name: string; input: unknown } + | { type: 'done'; text: string; usage: IUsage; iterations: number } + | { type: 'error'; error: string } + +const DEFAULT_OUTPUT_MAX_CHARS = 8_000 +const DEFAULT_MAX_CONCURRENT = 4 +const EVENT_STRING_MAX = 4_000 + +// JSON round-trip that turns Errors into their message and drops what JSON +// cannot carry. The fallback when structuredClone refuses a payload. +const jsonSafe = (value: unknown): unknown => { + try { + const s = JSON.stringify(value, (_k, v: unknown) => + v instanceof Error ? v.message : typeof v === 'bigint' ? v.toString() : v, + ) + return s === undefined ? undefined : JSON.parse(s) + } catch { + return String(value) + } +} + +// Structured-cloneable copy of a payload (worker messages). +export const toCloneable = (value: T): T => { + try { + return structuredClone(value) + } catch { + return jsonSafe(value) as T + } +} + +const clipStrings = (value: unknown, depth = 0): unknown => { + if (typeof value === 'string') { + return clipText(value, EVENT_STRING_MAX) + } + if (value instanceof Error) { + return value + } + if (!value || typeof value !== 'object' || depth > 8) { + return value + } + if (Array.isArray(value)) { + return value.map((v) => clipStrings(v, depth + 1)) + } + const out: Record = {} + for (const [k, v] of Object.entries(value)) { + out[k] = clipStrings(v, depth + 1) + } + return out +} + +// A child event as forwarded to the parent: verbatim, long strings clipped. +export const slimEvent = (event: AgentEvent): AgentEvent => clipStrings(event) as AgentEvent + +/** + * Throws naming the first function-valued field of a config that has to + * cross into a worker thread (functions are not structured-cloneable). + */ +export const assertWorkerSafeConfig = (config: IAgentConfig, label = 'config'): void => { + if (config.tools && Object.keys(config.tools).length) { + throw new Error( + `${label}.tools cannot cross into a worker thread; use the \`tools\` option of createSubagentTool to proxy host tools (or isolation: 'in-process')`, + ) + } + const seen = new Set() + const walk = (value: unknown, path: string): void => { + if (typeof value === 'function') { + throw new Error( + `${path} is a function and cannot cross into a worker thread; use the \`tools\` option to proxy host tools, or isolation: 'in-process'`, + ) + } + if (!value || typeof value !== 'object' || seen.has(value)) { + return + } + seen.add(value) + for (const [k, v] of Object.entries(value)) { + walk(v, `${path}.${k}`) + } + } + walk(config, label) +} + +// Source runs (tests, --experimental-strip-types) use the .ts entry; the +// bundles (dist/index.js, dist/index.cjs via tsup's import.meta shim, +// dist/cli.js) sit next to dist/subagent-worker.js. +export const subagentWorkerUrl = (): URL => { + const here = import.meta.url + return new URL(here.endsWith('.ts') ? './subagent-worker.ts' : './subagent-worker.js', here) +} + +// The parent's node flags minus the ones that describe ITS entry point +// (-e / -p / --input-type would make the worker misread its own file). +// Everything else - notably --experimental-strip-types for source runs - is +// inherited. +export const workerExecArgv = (argv: readonly string[] = process.execArgv): string[] => { + const out: string[] = [] + for (let i = 0; i < argv.length; i++) { + const a = argv[i] + if (['-e', '--eval', '-p', '--print', '--input-type'].includes(a)) { + i++ + continue + } + if (/^--(input-type|eval|print)=/.test(a)) { + continue + } + out.push(a) + } + return out +} + +const createSemaphore = (max: number) => { + let active = 0 + const queue: (() => void)[] = [] + return { + acquire: (signal?: AbortSignal): Promise => + new Promise((resolve, reject) => { + if (signal?.aborted) { + reject(signal.reason ?? new DOMException('Aborted', 'AbortError')) + return + } + const grant = (): void => { + signal?.removeEventListener('abort', onAbort) + active++ + resolve() + } + function onAbort(): void { + const i = queue.indexOf(grant) + if (i >= 0) { + queue.splice(i, 1) + } + reject(signal?.reason ?? new DOMException('Aborted', 'AbortError')) + } + if (active < max) { + grant() + return + } + signal?.addEventListener('abort', onAbort, { once: true }) + queue.push(grant) + }), + release: (): void => { + active-- + queue.shift()?.() + }, + } +} + +// Run a host tool for a proxied call; the output is made cloneable. +const runHostTool = async ( + host: Tool | undefined, + name: string, + input: unknown, + callId: string, + signal: AbortSignal, +): Promise<{ ok: boolean; output: unknown }> => { + if (!host?.execute) { + return { ok: false, output: `Unknown proxied tool "${name}"` } + } + try { + const options = { + toolCallId: callId, + messages: [], + abortSignal: signal, + context: {}, + } as unknown as ToolExecutionOptions + let output = (host.execute as (i: unknown, o: unknown) => unknown)(input, options) + if (output && typeof (output as AsyncIterable)[Symbol.asyncIterator] === 'function') { + let last: unknown + for await (const v of output as AsyncIterable) { + last = v + } + output = last + } + return { ok: true, output: jsonSafe(await output) } + } catch (err) { + return { ok: false, output: errorMessage(err) } + } +} + +const describeTools = (tools: ToolSet): IProxiedToolDescriptor[] => + Object.entries(tools).map(([name, t]) => ({ + name, + description: typeof t.description === 'string' ? t.description : '', + inputSchema: asSchema(t.inputSchema).jsonSchema, + ...(isReadOnlyTool(t) ? { readOnly: true } : {}), + })) + +interface IChildOutcome { + text: string + usage: IUsage +} + +const runInProcess = async ( + opts: ISubagentToolOptions, + task: string, + signal: AbortSignal, + onEvent: EventHandler, +): Promise => { + const tools = { ...(opts.config.tools ?? {}), ...(opts.tools ?? {}) } + const child = await createAgent( + { ...opts.config, ...(Object.keys(tools).length ? { tools } : {}) }, + onEvent, + ) + try { + const r = await child.run({ input: task, signal }) + return { text: r.text, usage: r.usage } + } finally { + await child.close({ timeoutMs: 5_000 }).catch(() => {}) + } +} + +const runInWorker = ( + opts: ISubagentToolOptions, + task: string, + signal: AbortSignal, + onEvent: EventHandler, +): Promise => + new Promise((resolve, reject) => { + const hostTools = opts.tools ?? {} + let settled = false + let worker: Worker + try { + worker = new Worker(subagentWorkerUrl(), { + workerData: { subagent: opts.name }, + execArgv: workerExecArgv(), + }) + } catch (err) { + reject(err) + return + } + const finish = (fn: () => void): void => { + if (settled) { + return + } + settled = true + signal.removeEventListener('abort', onAbort) + fn() + void worker.terminate().catch(() => {}) + } + function onAbort(): void { + worker.postMessage({ type: 'abort' } satisfies ParentToWorkerMessage) + // Give the child a moment to unwind cleanly, then cut it off. + const t = setTimeout( + () => finish(() => reject(signal.reason ?? new DOMException('Aborted', 'AbortError'))), + 1_000, + ) + t.unref?.() + } + // Messages arrive outside of the tool call's async context; bind the + // handler so proxied host tools still see the parent run (runId, + // sandbox, approval events). + const onMessage = AsyncResource.bind((msg: WorkerToParentMessage) => { + switch (msg.type) { + case 'event': + onEvent(msg.event) + break + case 'tool-call': + void runHostTool(hostTools[msg.name], msg.name, msg.input, msg.callId, signal).then( + (r) => { + if (!settled) { + worker.postMessage({ + type: 'tool-result', + callId: msg.callId, + ok: r.ok, + output: r.output, + } satisfies ParentToWorkerMessage) + } + }, + ) + break + case 'done': + finish(() => resolve({ text: msg.text, usage: msg.usage })) + break + case 'error': + finish(() => + reject( + signal.aborted + ? (signal.reason ?? new DOMException('Aborted', 'AbortError')) + : new Error(msg.error), + ), + ) + break + } + }) + worker.on('message', onMessage) + worker.on('error', (err) => finish(() => reject(err))) + worker.on('exit', (code) => + finish(() => reject(new Error(`subagent worker exited with code ${code} before finishing`))), + ) + if (signal.aborted) { + onAbort() + return + } + signal.addEventListener('abort', onAbort, { once: true }) + worker.postMessage({ + type: 'run', + task, + config: toCloneable(opts.config), + proxied: describeTools(hostTools), + } satisfies ParentToWorkerMessage) + }) + +/** + * Expose a whole agent as ONE tool of a parent agent: input `{ task }`, + * output the child's final text. Calls in one model step run in parallel + * (bounded by maxConcurrent); the parent's abort aborts the child; child + * usage is emitted on the parent as `usage` (phase 'subagent') so it counts + * against the parent's limits; child events arrive as `subagent.event`. + */ +export const createSubagentTool = (opts: ISubagentToolOptions): Tool => { + const isolation: SubagentIsolation = opts.isolation ?? 'worker' + if (!opts.name || typeof opts.name !== 'string') { + throw new Error('createSubagentTool: name is required') + } + if (isolation === 'worker') { + assertWorkerSafeConfig(opts.config, `subagent "${opts.name}": config`) + } + const outputMaxChars = opts.outputMaxChars ?? DEFAULT_OUTPUT_MAX_CHARS + const sem = createSemaphore(Math.max(1, opts.maxConcurrent ?? DEFAULT_MAX_CONCURRENT)) + + const subagent = tool({ + description: opts.description, + inputSchema: z.object({ + task: z + .string() + .describe('A complete, self-contained task for the subagent, with all context it needs'), + }), + execute: async ({ task }, { abortSignal }) => { + const store = runContext.getStore() + const emit: EventHandler = store?.emit ?? (() => {}) + const id = randomUUID() + const name = opts.name + await sem.acquire(abortSignal) + const ac = new AbortController() + const onParentAbort = (): void => ac.abort(abortSignal?.reason) + abortSignal?.addEventListener('abort', onParentAbort, { once: true }) + if (abortSignal?.aborted) { + onParentAbort() + } + let timedOut = false + const timer = + typeof opts.timeoutMs === 'number' && opts.timeoutMs > 0 + ? setTimeout(() => { + timedOut = true + ac.abort(new Error(`subagent "${name}" timed out after ${opts.timeoutMs}ms`)) + }, opts.timeoutMs) + : undefined + let childUsage: IUsage = emptyUsage() + const onChildEvent: EventHandler = (event) => { + if (event.type === 'usage') { + // Counted on the parent as it happens, so the parent's limits + // see a long-running child. + const usage = normalizeUsage(event.usage) + emit({ type: 'usage', phase: 'subagent', usage }) + } + emit({ type: 'subagent.event', id, name, event: slimEvent(event) }) + } + try { + emit({ type: 'subagent.start', id, name, task }) + const out = + isolation === 'worker' + ? await runInWorker(opts, task, ac.signal, onChildEvent) + : await runInProcess(opts, task, ac.signal, onChildEvent) + childUsage = normalizeUsage(out.usage) + const text = clipText(out.text, outputMaxChars) + emit({ type: 'subagent.complete', id, name, text, usage: childUsage }) + return text + } catch (err) { + const message = timedOut + ? `subagent "${name}" timed out after ${opts.timeoutMs}ms` + : errorMessage(err) + emit({ type: 'subagent.error', id, name, error: message }) + if (timedOut) { + throw new Error(message) + } + throw err + } finally { + if (timer) { + clearTimeout(timer) + } + abortSignal?.removeEventListener('abort', onParentAbort) + sem.release() + } + }, + }) + return markReadOnly(subagent, opts.readOnly === true) +} diff --git a/src/synthesizer.ts b/src/synthesizer.ts index 723b74d..dd505c7 100644 --- a/src/synthesizer.ts +++ b/src/synthesizer.ts @@ -1,9 +1,11 @@ import { streamText } from 'ai' +import { stageCallOptions } from './call-options.ts' import type { IAgentInternalContext } from './internal.ts' -import { SYNTHESIZER_SYSTEM, buildSynthesizerUserPrompt, withDomainContext } from './prompts.ts' +import { buildSynthesizerUserPrompt, composeSynthesizerSystem } from './prompts.ts' +import { stageEmitsThoughts } from './thinking.ts' import { ATTR, withSpan } from './tracing.ts' import type { IConversationTurn, IPlan, IStepResult, IUsage } from './types.ts' -import { withRetry, withTimeout } from './utils.ts' +import { normalizeUsage, withRetry, withTimeout } from './utils.ts' export const synthesizeAnswer = async ( input: string, @@ -36,10 +38,23 @@ const streamOnce = async ( ctx: IAgentInternalContext, signal?: AbortSignal, ): Promise<{ text: string; usage: IUsage }> => { + const call = stageCallOptions( + ctx.config, + 'synthesizer', + composeSynthesizerSystem({ + domain: ctx.config.systemPrompt, + skills: ctx.skills, + activeSkills: ctx.run?.activeSkills, + }), + ) + const view = ctx.run + ? { summary: ctx.run.traceSummary, upTo: ctx.run.traceSummaryUpTo } + : undefined + const emitThoughts = stageEmitsThoughts(ctx.config, 'synthesizer') const result = streamText({ model: ctx.synthesizerModel, - system: withDomainContext(SYNTHESIZER_SYSTEM, ctx.config.systemPrompt), - prompt: buildSynthesizerUserPrompt(input, plan, trace, history), + ...call, + prompt: buildSynthesizerUserPrompt(input, plan, trace, history, view), abortSignal: withTimeout(signal, ctx.config.llmTimeoutMs ?? 0), }) @@ -48,6 +63,10 @@ const streamOnce = async ( if (part.text) { ctx.emit({ type: 'final.text-delta', delta: part.text }) } + } else if (part.type === 'reasoning-delta') { + if (part.text && emitThoughts) { + ctx.emit({ type: 'final.reasoning-delta', delta: part.text }) + } } else if (part.type === 'error') { const error = part.error throw error instanceof Error ? error : new Error(String(error)) @@ -55,12 +74,5 @@ const streamOnce = async ( } const [text, rawUsage] = await Promise.all([result.text, result.usage]) - return { - text, - usage: { - inputTokens: rawUsage.inputTokens ?? 0, - outputTokens: rawUsage.outputTokens ?? 0, - totalTokens: rawUsage.totalTokens ?? 0, - }, - } + return { text, usage: normalizeUsage(rawUsage) } } diff --git a/src/thinking.ts b/src/thinking.ts new file mode 100644 index 0000000..bafa644 --- /dev/null +++ b/src/thinking.ts @@ -0,0 +1,139 @@ +import type { + AgentStage, + IAgentConfig, + IThinkingConfig, + ThinkingLevel, + ThinkingSetting, +} from './types.ts' + +export type ProviderOptionsMap = Record> + +export interface IResolvedThinking { + // Portable AI SDK `reasoning` call option. + reasoning?: ThinkingLevel + // Provider-specific call options (budgets, "stream the thoughts"). + providerOptions?: ProviderOptionsMap +} + +const LEVELS: ReadonlySet = new Set([ + 'provider-default', + 'none', + 'minimal', + 'low', + 'medium', + 'high', + 'xhigh', +]) + +export const isThinkingLevel = (v: unknown): v is ThinkingLevel => + typeof v === 'string' && LEVELS.has(v) + +// Normalise any ThinkingSetting to a config object, or undefined for "send +// nothing" (false / undefined). +export const normalizeThinking = ( + setting: ThinkingSetting | undefined, +): IThinkingConfig | undefined => { + if (setting === undefined || setting === false) { + return undefined + } + if (setting === true) { + return { level: 'medium' } + } + if (typeof setting === 'string') { + return { level: setting } + } + return { ...setting, level: setting.level ?? 'medium' } +} + +/** + * Map a thinking setting to AI SDK call options. Pure. + * + * - false / undefined -> {} (provider default, nothing sent) + * - 'none' -> { reasoning: 'none' } (explicit disable, no provider options) + * - budgetTokens -> Anthropic `thinking` + Google `thinkingConfig` budgets + * - otherwise, unless includeThoughts is false -> ask Google / OpenAI to + * return their thoughts (summaries) so they can be streamed. + * + * Provider-specific options take precedence over `reasoning` inside the SDK. + */ +export const resolveThinking = (setting: ThinkingSetting | undefined): IResolvedThinking => { + const cfg = normalizeThinking(setting) + if (!cfg) { + return {} + } + const level: ThinkingLevel = cfg.level ?? 'medium' + if (level === 'none') { + return { reasoning: 'none' } + } + const includeThoughts = cfg.includeThoughts !== false + const budget = cfg.budgetTokens + if (typeof budget === 'number' && Number.isFinite(budget) && budget > 0) { + const budgetTokens = Math.floor(budget) + return { + reasoning: level, + providerOptions: { + anthropic: { thinking: { type: 'enabled', budgetTokens } }, + google: { thinkingConfig: { thinkingBudget: budgetTokens, includeThoughts } }, + }, + } + } + if (includeThoughts) { + return { + reasoning: level, + providerOptions: { + google: { thinkingConfig: { includeThoughts: true } }, + openai: { reasoningSummary: 'auto' }, + }, + } + } + return { reasoning: level } +} + +// The effective setting of one stage: a stageThinking entry (even `false`) +// wins over the top-level `thinking`. +export const stageThinkingSetting = ( + config: Pick, + stage: AgentStage, +): ThinkingSetting | undefined => { + const own = config.stageThinking?.[stage] + return own !== undefined ? own : config.thinking +} + +// Whether a stage wants its thoughts streamed (reasoning-delta events). +export const stageEmitsThoughts = ( + config: Pick, + stage: AgentStage, +): boolean => normalizeThinking(stageThinkingSetting(config, stage))?.includeThoughts !== false + +const isPlainObject = (v: unknown): v is Record => + Boolean(v) && typeof v === 'object' && !Array.isArray(v) + +const deepMerge = (a: Record, b: Record) => { + const out: Record = { ...a } + for (const [k, v] of Object.entries(b)) { + out[k] = isPlainObject(out[k]) && isPlainObject(v) ? deepMerge(out[k], v) : v + } + return out +} + +// Deep-merge call-level provider options per provider key; later layers win. +// Returns undefined when every layer is empty, so callers can spread it +// without sending an empty object. +export const mergeProviderOptions = ( + ...layers: (ProviderOptionsMap | undefined)[] +): ProviderOptionsMap | undefined => { + let out: ProviderOptionsMap | undefined + for (const layer of layers) { + if (!layer) { + continue + } + for (const [provider, opts] of Object.entries(layer)) { + if (!isPlainObject(opts)) { + continue + } + out ??= {} + out[provider] = deepMerge(out[provider] ?? {}, opts) + } + } + return out +} diff --git a/src/tool-search.ts b/src/tool-search.ts new file mode 100644 index 0000000..f0205de --- /dev/null +++ b/src/tool-search.ts @@ -0,0 +1,184 @@ +import { tool, type ToolSet } from 'ai' +import { z } from 'zod' +import { markReadOnly } from './approval.ts' +import { runContext } from './context.ts' +import type { EffectiveToolStrategy } from './internal.ts' +import type { IAgentConfig, IToolCatalogEntry } from './types.ts' + +export const FIND_TOOLS_TOOL = 'find_tools' +export const DEFAULT_TOOL_SEARCH_THRESHOLD = 40 +// Discovered tools carried into later steps (most recent first). +export const MAX_CARRIED_DISCOVERED_TOOLS = 16 +const DEFAULT_SEARCH_LIMIT = 8 +const MAX_SEARCH_LIMIT = 20 + +// 'auto' -> 'all' up to toolSearchThreshold tools, 'search' above. +export const resolveToolStrategy = ( + config: Pick, + catalogSize: number, +): EffectiveToolStrategy => { + const s = config.toolSelectionStrategy ?? 'auto' + if (s === 'auto') { + const threshold = config.toolSearchThreshold ?? DEFAULT_TOOL_SEARCH_THRESHOLD + return catalogSize > threshold ? 'search' : 'all' + } + return s +} + +// Lowercased word tokens; splits on non-alphanumerics AND camelCase +// ("listOpenIssues" -> list, open, issues; "github__get_PR" -> github, get, pr). +export const tokenizeForSearch = (text: string): string[] => + text + .replace(/([a-z0-9])([A-Z])/g, '$1 $2') + .replace(/([A-Z]+)([A-Z][a-z])/g, '$1 $2') + .toLowerCase() + .split(/[^a-z0-9]+/) + .filter(Boolean) + +const hit = (tokens: string[], term: string): boolean => + tokens.some((t) => t === term || (term.length >= 3 && t.startsWith(term))) + +export interface ISearchToolsOptions { + // Only tools of this MCP server ('' for config.tools). + server?: string + // Default 8, clamped to 1..20. + limit?: number +} + +/** + * Rank catalogue entries for a free-text query. Pure. + * score = Σ over query terms of 3·nameHit + 2·serverHit + 1·descriptionHit, + * with prefix matching for terms of 3+ chars. Zero scores are dropped; ties + * keep catalogue order. + */ +export const searchTools = ( + catalog: T[], + query: string, + opts: ISearchToolsOptions = {}, +): T[] => { + const terms = [...new Set(tokenizeForSearch(query))] + if (terms.length === 0) { + return [] + } + const raw = Math.floor(opts.limit ?? DEFAULT_SEARCH_LIMIT) + const limit = Math.min(MAX_SEARCH_LIMIT, Math.max(1, Number.isFinite(raw) ? raw : 8)) + const server = opts.server?.trim().toLowerCase() + const scored: { entry: T; score: number; index: number }[] = [] + catalog.forEach((entry, index) => { + if (server && (entry.server ?? '').toLowerCase() !== server) { + return + } + const nameTokens = tokenizeForSearch(entry.name) + const serverTokens = tokenizeForSearch(entry.server ?? '') + const descTokens = tokenizeForSearch(entry.description ?? '') + let score = 0 + for (const term of terms) { + score += + (hit(nameTokens, term) ? 3 : 0) + + (hit(serverTokens, term) ? 2 : 0) + + (hit(descTokens, term) ? 1 : 0) + } + if (score > 0) { + scored.push({ entry, score, index }) + } + }) + scored.sort((a, b) => b.score - a.score || a.index - b.index) + return scored.slice(0, limit).map((s) => s.entry) +} + +export const SEARCH_CATALOG_BUDGET_CHARS = 12_000 +const SEARCH_DESC_CHARS = 60 + +/** + * The planner's / replanner's view of a large catalogue: grouped by server, + * one short line per tool, within a char budget, then a pointer to + * find_tools for the rest. + */ +export const renderSearchCatalog = ( + catalog: IToolCatalogEntry[], + budget = SEARCH_CATALOG_BUDGET_CHARS, +): string => { + if (catalog.length === 0) { + return '(no tools available)' + } + const groups = new Map() + for (const entry of catalog) { + const key = entry.server ?? '' + const list = groups.get(key) ?? [] + list.push(entry) + groups.set(key, list) + } + const lines: string[] = [] + let used = 0 + let shown = 0 + outer: for (const [server, entries] of groups) { + const header = `[${server}]` + if (used + header.length + 1 > budget) { + break + } + lines.push(header) + used += header.length + 1 + for (const t of entries) { + const desc = (t.description || '').replace(/\s+/g, ' ').trim().slice(0, SEARCH_DESC_CHARS) + const line = `- ${t.name}: ${desc}` + if (used + line.length + 1 > budget) { + break outer + } + lines.push(line) + used += line.length + 1 + shown++ + } + } + const rest = catalog.length - shown + if (rest > 0) { + lines.push(`… ${rest} more tools — the executor can find them with ${FIND_TOOLS_TOOL}`) + } + return lines.join('\n') +} + +/** + * The find_tools built-in: searches the live catalogue and ACTIVATES the + * matches for the rest of the executor call and the run. + */ +export const createFindToolsTool = (getCatalog: () => IToolCatalogEntry[]): ToolSet => ({ + [FIND_TOOLS_TOOL]: markReadOnly( + tool({ + description: + 'Search the full tool catalogue by keywords and ACTIVATE the matching tools: they become callable from your next step on. Use it whenever the tools you need are not loaded yet.', + inputSchema: z.object({ + query: z + .string() + .describe('Keywords describing the capability you need, e.g. "list github issues"'), + server: z.string().optional().describe('Restrict the search to one MCP server'), + limit: z.number().optional().describe('Max results (default 8, max 20)'), + }), + execute: async ({ query, server, limit }) => { + const found = searchTools(getCatalog(), query, { server, limit }) + const names = found.map((t) => t.name) + const store = runContext.getStore() + const state = store?.state + if (state) { + for (const name of names) { + const at = state.discovered.indexOf(name) + if (at >= 0) { + state.discovered.splice(at, 1) + } + state.discovered.push(name) + state.stepActive?.add(name) + } + } + if (store?.emit && store.currentStep) { + store.emit({ type: 'tools.discovered', step: store.currentStep, query, names }) + } + return { + tools: found.map((t) => ({ + name: t.name, + description: t.description, + server: t.server ?? '', + readOnly: t.readOnly === true, + })), + } + }, + }), + ), +}) diff --git a/src/tool-wrap.ts b/src/tool-wrap.ts new file mode 100644 index 0000000..0afebe7 --- /dev/null +++ b/src/tool-wrap.ts @@ -0,0 +1,110 @@ +import type { Tool, ToolExecutionOptions, ToolSet } from 'ai' +import { isReadOnlyTool, type IApprovalController } from './approval.ts' +import { withToolOutputLimit } from './compaction.ts' +import { runContext } from './context.ts' + +export interface IToolWrapDeps { + approval: IApprovalController + // Cap on tool calls per run (built-ins do not count). + maxToolCalls?: number + // Model-visible output cap per result (0 = unlimited). + maxToolOutputChars: number +} + +const WRAPPED = Symbol.for('@dudko.dev/agent:wrapped-tool') + +export class ToolBudgetError extends Error { + constructor(cap: number) { + super( + `tool-call budget exhausted (maxToolCalls=${cap}). Do not call more tools; finish the step with what you have.`, + ) + this.name = 'ToolBudgetError' + } + + toJSON(): string { + return this.message + } +} + +// Reserve one call of the run's tool budget; returns a release for a call +// that ends up not running (denied). +const reserveToolCall = (cap: number | undefined): (() => void) => { + const state = runContext.getStore()?.state + if (!state) { + return () => {} + } + if (typeof cap === 'number' && cap > 0 && state.toolCalls >= cap) { + throw new ToolBudgetError(cap) + } + state.toolCalls++ + return () => { + state.toolCalls-- + } +} + +const isAsyncGeneratorFunction = (fn: unknown): boolean => + Object.prototype.toString.call(fn) === '[object AsyncGeneratorFunction]' + +/** + * Wrap one tool's execute with the run-level gates — tool-call budget, then + * the approval gate — and cap what the model sees of its results. The gate + * runs INSIDE execute, so it works for every tool (MCP, native, subagent, + * built-in) and every call path. Idempotent. + */ +export const wrapTool = ( + name: string, + tool: Tool, + deps: IToolWrapDeps, + opts: { builtIn?: boolean } = {}, +): Tool => { + if ((tool as { [WRAPPED]?: boolean })[WRAPPED]) { + return tool + } + const readOnly = isReadOnlyTool(tool) + const builtIn = opts.builtIn === true + const inner = tool.execute as + ((input: unknown, options: ToolExecutionOptions) => unknown) | undefined + const before = async (input: unknown, options: ToolExecutionOptions) => { + const release = builtIn ? () => {} : reserveToolCall(deps.maxToolCalls) + try { + await deps.approval.gate(name, input, { + readOnly, + builtIn, + signal: options?.abortSignal, + }) + } catch (err) { + release() + throw err + } + } + let execute: unknown + if (inner) { + execute = isAsyncGeneratorFunction(inner) + ? async function* (input: unknown, options: ToolExecutionOptions) { + await before(input, options) + yield* inner(input, options) as AsyncIterable + } + : async (input: unknown, options: ToolExecutionOptions) => { + await before(input, options) + return inner(input, options) + } + } + const wrapped = withToolOutputLimit( + { ...tool, ...(execute ? { execute } : {}) } as Tool, + deps.maxToolOutputChars, + ) + Object.defineProperty(wrapped, WRAPPED, { value: true, enumerable: false }) + return wrapped +} + +export const wrapToolSet = ( + tools: ToolSet, + deps: IToolWrapDeps, + opts: { builtIn?: boolean } = {}, +): ToolSet => { + const out: ToolSet = {} + for (const [name, tool] of Object.entries(tools)) { + out[name] = wrapTool(name, tool, deps, opts) + } + return out +} diff --git a/src/types.ts b/src/types.ts index 5e05dad..d38021e 100644 --- a/src/types.ts +++ b/src/types.ts @@ -18,11 +18,170 @@ export type ProviderType = export type LogLevel = 'none' | 'error' | 'warn' | 'info' | 'debug' // 'all' - executor receives the full filtered ToolSet on every step. -// Best for catalogs <= ~50 tools. +// Best for catalogs <= ~40 tools. // 'plan-narrowed' - executor receives only tools listed in step.suggestedTools. // Planner is required to populate suggestedTools when a step // needs tools; empty means "reasoning-only step". -export type ToolSelectionStrategy = 'all' | 'plan-narrowed' +// 'search' - executor starts each step with the built-in tools, the +// step's suggestedTools and the tools discovered earlier in +// the run, plus `find_tools`, which searches the whole +// catalogue and activates matches for the rest of the run. +// Built for catalogues of hundreds of tools. +// 'auto' (default) - 'all' while the filtered catalogue has at most +// `toolSearchThreshold` (default 40) tools, 'search' above. +export type ToolSelectionStrategy = 'all' | 'plan-narrowed' | 'search' | 'auto' + +// The four LLM stages of the loop. The replanner shares the planner's model +// but has its own thinking / per-call budget knobs. +export type AgentStage = 'planner' | 'executor' | 'replanner' | 'synthesizer' + +// Portable reasoning effort, mapped by the AI SDK onto each provider's own +// knob. 'none' explicitly disables thinking; 'provider-default' sends the +// provider's default level. +export type ThinkingLevel = + 'provider-default' | 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' + +export interface IThinkingConfig { + // Default 'medium'. + level?: ThinkingLevel + // Exact token budget. Becomes provider-specific options for providers that + // take a budget rather than an effort level (Anthropic, Google). + budgetTokens?: number + // Ask the provider to stream its thoughts (step.reasoning-delta / + // final.reasoning-delta) where it supports that. Default true. + includeThoughts?: boolean +} + +// false / undefined -> send nothing (provider default); true -> 'medium'; +// a bare level -> { level }. +export type ThinkingSetting = boolean | ThinkingLevel | IThinkingConfig + +// Cumulative per-run token caps. Soft caps: checked between steps and at +// every executor LLM step boundary; once crossed the run stops executing +// steps and goes straight to synthesis (which still runs, capped by +// perCall.synthesizer). +export interface ITokenLimits { + maxInputTokens?: number + // Includes reasoning tokens. + maxOutputTokens?: number + maxReasoningTokens?: number + // input + output. Falls back to the legacy top-level `maxTotalTokens`. + maxTotalTokens?: number + // maxOutputTokens for every single LLM call of a stage. + perCall?: { + planner?: number + executor?: number + replanner?: number + synthesizer?: number + compaction?: number + } +} + +export type TokenLimitKind = 'input' | 'output' | 'reasoning' | 'total' + +// What a `budget.exceeded` event reports. For 'tool-calls', `tokens` carries +// the number of tool calls made. +export type BudgetKind = TokenLimitKind | 'tool-calls' + +export interface ILimitBreach { + kind: TokenLimitKind + tokens: number + cap: number +} + +// Context compaction. Everything has a default; pass `{ auto: false }` to +// keep only the manual `agent.compact()` and the tool-output cap. +export interface ICompactionConfig { + // Compact history (at run start) and trace (before each executor / + // replanner / synthesizer call) when their estimate crosses the threshold. + // Default true. + auto?: boolean + // Model context window, in tokens. Default 128_000. + contextWindowTokens?: number + // Compact when the estimated context exceeds this. Default: 50% of the + // window. + thresholdTokens?: number + // History turns kept verbatim (default 4). + keepRecentTurns?: number + // Trace steps kept verbatim (default 3). + keepRecentSteps?: number + // Output cap of a summary call (default 1024). + summaryMaxTokens?: number + // Per tool result the MODEL sees (default 20_000; 0 = unlimited). The raw + // output still lands in the trace and the events. + maxToolOutputChars?: number + // Inside one step's tool loop: once its context passes this many tokens, + // the oldest tool results become one-line stubs (context editing). Default: + // a quarter of the window; 0 = never. + clearToolResultsAfterTokens?: number + // Most recent tool results kept verbatim when clearing (default 3). + keepToolResults?: number +} + +// An agentskills.io-style skill: a name + a one-line description shown in an +// index, and instructions the agent loads only when the skill applies. +export interface ISkill { + // [a-z0-9-]{1,64} + name: string + // When to use it (shown in the index). + description: string + // Full instructions (markdown body). + content: string + // Bundled text resources, skill-relative POSIX paths. + files?: { path: string; content: string }[] +} + +// How tool calls are consented: +// 'autopilot' - every call runs (default); +// 'ask-writes' - read-only tools run, anything else asks; +// 'ask-all' - every call asks; +// 'read-only' - read-only tools run, anything else is denied. +export type ToolApprovalMode = 'autopilot' | 'ask-writes' | 'ask-all' | 'read-only' + +export type ToolPermission = 'allow' | 'ask' | 'deny' + +export interface IToolApprovalRequest { + // Unique per request. + id: string + toolName: string + // The tool input AFTER inputSanitizer. + input: unknown + // The tool is known to be read-only (MCP readOnlyHint or markReadOnly). + readOnly: boolean + step?: IPlanStep + runId?: string +} + +// `remember: true` on an approval allows the tool for the rest of this agent +// instance's life without asking again. +export type ToolApprovalDecision = + boolean | { approved: boolean; reason?: string; remember?: boolean } + +export interface IToolApprovalConfig { + // Default 'autopilot'. Switch at runtime with agent.setToolApprovalMode(). + mode?: ToolApprovalMode + // Exact tool names or '*' globs, e.g. { 'github__delete_*': 'deny' }. + // The most specific match wins (exact > longest glob) and beats the mode. + rules?: Record + // Called for every 'ask'. Without it an 'ask' is a denial. + onRequest?: (req: IToolApprovalRequest) => ToolApprovalDecision | Promise + // No decision within this many ms -> deny. Default: wait forever. + timeoutMs?: number +} + +// true (default) / false, or options: Anthropic cache TTL and the OpenAI +// prompt cache key (default `${clientName}:${stage}`). +export type PromptCachingSetting = boolean | { ttl?: '5m' | '1h'; key?: string } + +// One entry of the tool catalogue (agent.listTools()). +export interface IToolCatalogEntry { + name: string + description: string + // MCP server name, or '' for config.tools. + server?: string + // Known to be read-only (MCP annotations.readOnlyHint, or markReadOnly()). + readOnly?: boolean +} // Remote MCP server reached over StreamableHTTP. (Legacy HTTP+SSE servers - // the 2024-11-05 transport with a separate /sse endpoint - are NOT supported; @@ -44,6 +203,10 @@ export interface IMcpHttpServerConfig { // Custom fetch for every HTTP request the transport and the OAuth flow make: // corporate proxies, mTLS agents, instrumentation. fetch?: FetchLike + // A server that does not finish connect + tools/list within this many ms + // is reported as failed and closed; the others still mount. Overrides the + // top-level mcpConnectTimeoutMs. Default 30_000; 0 disables. + connectTimeoutMs?: number } // Local MCP server spawned as a child process. The transport speaks JSON-RPC @@ -54,6 +217,8 @@ export interface IMcpStdioServerConfig { args?: string[] env?: Record cwd?: string + // See IMcpHttpServerConfig.connectTimeoutMs. + connectTimeoutMs?: number } // Discriminated union: callers pick HTTP or stdio per server. Existing @@ -139,8 +304,35 @@ export interface IAgentConfig { // predicate can never stall the run. replanAfter?: ReplanTrigger // Soft cap on cumulative tokens; checked between steps and triggers an - // early jump to synthesis when crossed. + // early jump to synthesis when crossed. Legacy shortcut for + // `limits.maxTotalTokens` (which wins when both are set). maxTotalTokens?: number + // Cumulative per-run token caps + per-call output caps. See ITokenLimits. + limits?: ITokenLimits + // Cap on tool calls per run (across steps). Once reached, further calls + // fail with "tool-call budget exhausted" and the run goes to synthesis. + maxToolCalls?: number + // Hard cap on the number of steps in a plan (initial and revised). + // Default 8. + maxPlanSteps?: number + // Thinking / reasoning for every stage. Per-stage entries in stageThinking + // win. Compaction calls never think. + thinking?: ThinkingSetting + stageThinking?: Partial> + // History / trace compaction and the model-visible tool-output cap. + compaction?: ICompactionConfig + // Skills the planner can pick and the executor can load on demand. + skills?: ISkill[] + // Tool-call consent. Default: autopilot (every call runs). + toolApproval?: IToolApprovalConfig + // 'auto' switches to 'search' above this many tools. Default 40. + toolSearchThreshold?: number + // Default connect + tools/list timeout for every MCP server (each server + // can override it). Default 30_000; 0 disables. + mcpConnectTimeoutMs?: number + // Provider prompt caching. Default true: run-stable system prompts, an + // Anthropic cache breakpoint on them, and an OpenAI prompt cache key. + promptCaching?: PromptCachingSetting llmTimeoutMs?: number llmMaxRetries?: number // Hard cap on the number of concurrent agent.run() calls a single agent @@ -194,6 +386,13 @@ export interface IUsage { inputTokens: number outputTokens: number totalTokens: number + // Optional in the type for back-compat; the agent always fills them. + // Part of outputTokens. + reasoningTokens?: number + // Input tokens served from the provider's prompt cache (part of inputTokens). + cachedInputTokens?: number + // Input tokens written to the provider's prompt cache. + cacheWriteTokens?: number } export interface IPlanStep { @@ -212,6 +411,9 @@ export interface IPlanStep { export interface IPlan { thought: string steps: IPlanStep[] + // Names of configured skills the planner picked for this request. Their + // instructions are injected into the executor / replanner / synthesizer. + skills?: string[] } export interface IStepResult { @@ -230,6 +432,11 @@ export type ReplanTrigger = export type ReplanCause = 'last-step' | 'clean-step' | 'llm-decision' +// Phase of a `usage` event. 'compact' is a compaction summary call; +// 'subagent' is usage a subagent tool spent (it counts against this run's +// limits like any other). +export type UsagePhase = 'plan' | 'execute' | 'replan' | 'synthesize' | 'compact' | 'subagent' + type AgentEventBody = | { type: 'plan.thought-delta'; delta: string } // Fires once per planner step as soon as the structured-output stream has @@ -241,6 +448,8 @@ type AgentEventBody = | { type: 'plan.revised'; plan: IPlan; reason: string } | { type: 'step.start'; step: IPlanStep; index: number } | { type: 'step.text-delta'; step: IPlanStep; delta: string } + // The model's thoughts, when thinking is on and the provider streams them. + | { type: 'step.reasoning-delta'; step: IPlanStep; delta: string } | { type: 'step.tool-call'; step: IPlanStep; name: string; input: unknown } | { type: 'step.tool-result'; step: IPlanStep; name: string; output: unknown; ok: boolean } | { type: 'step.complete'; step: IPlanStep; result: IStepResult } @@ -251,16 +460,53 @@ type AgentEventBody = cause: ReplanCause } | { type: 'final.text-delta'; delta: string } + | { type: 'final.reasoning-delta'; delta: string } | { type: 'final'; text: string } | { type: 'log'; level: LogLevel; message: string } - | { type: 'usage'; phase: 'plan' | 'execute' | 'replan' | 'synthesize'; usage: IUsage } + | { type: 'usage'; phase: UsagePhase; usage: IUsage } | { type: 'retry' phase: 'plan' | 'execute' | 'replan' | 'synthesize' attempt: number error: string } - | { type: 'budget.exceeded'; tokens: number; cap: number } + // A run-level cap was crossed; the run stops executing steps and + // synthesizes. For kind 'tool-calls', `tokens` is the tool-call count. + | { type: 'budget.exceeded'; kind: BudgetKind; tokens: number; cap: number } + | { + type: 'context.compacted' + // tool-results = stale results inside a step's tool loop. + scope: 'history' | 'trace' | 'tool-results' + beforeTokens: number + afterTokens: number + } + | { type: 'skill.activated'; name: string; by: 'plan' | 'tool' } + // find_tools found and activated these tools (search strategy). + | { type: 'tools.discovered'; step: IPlanStep; query: string; names: string[] } + // Only when the agent actually asks (onRequest is called). + | { + type: 'tool.approval-requested' + id: string + name: string + input: unknown + readOnly: boolean + step?: IPlanStep + } + // automatic: true for rule / mode / timeout denials nobody decided on. + // Automatic ALLOWs emit nothing, to keep the stream quiet. + | { + type: 'tool.approval-resolved' + id: string + name: string + approved: boolean + reason?: string + automatic: boolean + } + | { type: 'subagent.start'; id: string; name: string; task: string } + // A child event, forwarded verbatim (long strings clipped). + | { type: 'subagent.event'; id: string; name: string; event: AgentEvent } + | { type: 'subagent.complete'; id: string; name: string; text: string; usage: IUsage } + | { type: 'subagent.error'; id: string; name: string; error: string } | { type: 'revisions.exceeded'; cap: number } | { type: 'error'; error: Error; phase: 'plan' | 'execute' | 'replan' | 'synthesize' | 'init' } @@ -304,6 +550,9 @@ export interface IAgentRunResult { trace: IStepResult[] iterations: number usage: IUsage + // Set when the run compacted the history it was given (the caller's array + // is never mutated). Persist it in place of the old history. + compactedHistory?: IConversationTurn[] } // Snapshot of a run handed to IPersistence hooks. Each hook receives the @@ -337,6 +586,14 @@ export interface IRunSnapshot { text?: string error?: string completedAt?: number + // Running summary of trace[0, traceSummaryUpTo) once trace compaction has + // kicked in; prompts render it in place of those steps. + traceSummary?: string + traceSummaryUpTo?: number + // Skills activated so far (by the plan or by load_skill). + activeSkills?: string[] + // Tool calls made so far (counts against maxToolCalls on resume). + toolCallCount?: number } // Optional persistence facade. Write hooks fire at run start, after each diff --git a/src/utils.ts b/src/utils.ts index b260e14..d389a64 100644 --- a/src/utils.ts +++ b/src/utils.ts @@ -1,3 +1,5 @@ +import type { IUsage } from './types.ts' + export interface IRetryOptions { maxRetries: number baseDelayMs?: number @@ -113,3 +115,69 @@ export const redactHeaders = ( } return out } + +// A zeroed usage record with every optional detail filled in. +export const emptyUsage = (): Required => ({ + inputTokens: 0, + outputTokens: 0, + totalTokens: 0, + reasoningTokens: 0, + cachedInputTokens: 0, + cacheWriteTokens: 0, +}) + +// The subset of the AI SDK's LanguageModelUsage we read. Every field is +// optional so a provider that reports nothing (or a v6-shaped object) still +// normalises to zeros instead of NaN. +export interface ISdkUsageLike { + inputTokens?: number + outputTokens?: number + totalTokens?: number + inputTokenDetails?: { cacheReadTokens?: number; cacheWriteTokens?: number } + outputTokenDetails?: { reasoningTokens?: number } + // Pre-v6 flat fields, still produced by some wrappers. + reasoningTokens?: number + cachedInputTokens?: number +} + +const n = (v: unknown): number => (typeof v === 'number' && Number.isFinite(v) ? v : 0) + +// AI SDK usage -> IUsage with every detail field filled (0 when unknown). +export const normalizeUsage = (usage: ISdkUsageLike | undefined | null): Required => { + if (!usage) { + return emptyUsage() + } + const inputTokens = n(usage.inputTokens) + const outputTokens = n(usage.outputTokens) + return { + inputTokens, + outputTokens, + totalTokens: + typeof usage.totalTokens === 'number' ? usage.totalTokens : inputTokens + outputTokens, + reasoningTokens: n(usage.outputTokenDetails?.reasoningTokens ?? usage.reasoningTokens), + cachedInputTokens: n(usage.inputTokenDetails?.cacheReadTokens ?? usage.cachedInputTokens), + cacheWriteTokens: n(usage.inputTokenDetails?.cacheWriteTokens), + } +} + +// a + b, field by field (missing optional fields count as 0). +export const addUsage = (a: IUsage, b: IUsage): Required => ({ + inputTokens: a.inputTokens + b.inputTokens, + outputTokens: a.outputTokens + b.outputTokens, + totalTokens: a.totalTokens + b.totalTokens, + reasoningTokens: (a.reasoningTokens ?? 0) + (b.reasoningTokens ?? 0), + cachedInputTokens: (a.cachedInputTokens ?? 0) + (b.cachedInputTokens ?? 0), + cacheWriteTokens: (a.cacheWriteTokens ?? 0) + (b.cacheWriteTokens ?? 0), +}) + +// Add b into a, in place (the run accumulator is shared by reference). +export const accumulateUsage = (into: IUsage, b: IUsage): void => { + Object.assign(into, addUsage(into, b)) +} + +// Clip a string to max chars with a "… [truncated N chars]" marker. +export const clipText = (s: string, max: number): string => + max > 0 && s.length > max ? `${s.slice(0, max)}… [truncated ${s.length - max} chars]` : s + +export const errorMessage = (err: unknown): string => + err instanceof Error ? err.message : typeof err === 'string' ? err : JSON.stringify(err) diff --git a/tests/approval.test.ts b/tests/approval.test.ts new file mode 100644 index 0000000..cfafbab --- /dev/null +++ b/tests/approval.test.ts @@ -0,0 +1,192 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + createApprovalController, + decideToolPermission, + isReadOnlyTool, + markReadOnly, + matchToolRule, + ToolDeniedError, +} from '../src/approval.ts' +import { runContext } from '../src/context.ts' +import type { AgentEvent, IToolApprovalRequest } from '../src/types.ts' + +test('matchToolRule: exact beats any glob; the longest glob wins among globs', () => { + const rules = { + '*': 'ask', + 'github__*': 'allow', + 'github__delete_*': 'deny', + github__delete_repo: 'ask', + } as const + assert.deepEqual(matchToolRule(rules, 'github__delete_repo'), { + pattern: 'github__delete_repo', + permission: 'ask', + }) + assert.equal(matchToolRule(rules, 'github__delete_issue')?.permission, 'deny') + assert.equal(matchToolRule(rules, 'github__list')?.permission, 'allow') + assert.equal(matchToolRule(rules, 'jira__x')?.pattern, '*') + assert.equal(matchToolRule({ 'a.b*': 'deny' }, 'aXb'), undefined, 'regex chars are literal') + assert.equal(matchToolRule(undefined, 'x'), undefined) +}) + +test('decideToolPermission: mode defaults', () => { + const d = (mode: 'autopilot' | 'ask-writes' | 'ask-all' | 'read-only', readOnly: boolean) => + decideToolPermission({ name: 't', readOnly, mode }).permission + assert.equal(d('autopilot', false), 'allow') + assert.equal(d('read-only', true), 'allow') + assert.equal(d('read-only', false), 'deny') + assert.equal(d('ask-writes', true), 'allow') + assert.equal(d('ask-writes', false), 'ask') + assert.equal(d('ask-all', true), 'ask') +}) + +test('decideToolPermission order: built-in > remembered > rule > mode', () => { + const base = { name: 'fs__rm', readOnly: false, mode: 'ask-all' as const } + assert.deepEqual(decideToolPermission({ ...base, builtIn: true, rules: { '*': 'deny' } }), { + permission: 'allow', + source: 'builtin', + }) + assert.equal( + decideToolPermission({ ...base, remembered: new Set(['fs__rm']), rules: { 'fs__*': 'deny' } }) + .source, + 'remembered', + ) + assert.deepEqual(decideToolPermission({ ...base, rules: { 'fs__*': 'deny' } }), { + permission: 'deny', + source: 'rule', + rule: 'fs__*', + }) + // A rule beats the mode in both directions. + assert.equal( + decideToolPermission({ ...base, mode: 'read-only', rules: { fs__rm: 'allow' } }).permission, + 'allow', + ) + assert.equal( + decideToolPermission({ ...base, mode: 'autopilot', rules: { fs__rm: 'ask' } }).permission, + 'ask', + ) +}) + +test('markReadOnly / isReadOnlyTool', () => { + const t = markReadOnly({ description: 'x' }) + assert.equal(isReadOnlyTool(t), true) + assert.equal(isReadOnlyTool({}), false) + assert.equal(isReadOnlyTool(markReadOnly({}, false)), false) +}) + +test('ToolDeniedError carries the reason and serialises to its message', () => { + const e = new ToolDeniedError('not today') + assert.equal(e.name, 'ToolDeniedError') + assert.equal( + e.message, + 'Tool call denied by the user: not today. Do not retry it; continue without it, or report what is blocked.', + ) + assert.equal(JSON.parse(JSON.stringify({ e })).e, e.message) + assert.match(new ToolDeniedError().message, /^Tool call denied by the user\. Do not retry/) +}) + +const inRun = (events: AgentEvent[], fn: () => Promise): Promise => + runContext.run( + { + runId: 'run-1', + startedAt: 0, + sandboxDir: '/tmp/x', + emit: (e) => events.push(e), + currentStep: { id: 's1', description: 'd', expectedOutcome: 'e' }, + }, + fn, + ) + +test('approval gate: ask -> onRequest; remember skips later asks; events in order', async () => { + const requests: IToolApprovalRequest[] = [] + const events: AgentEvent[] = [] + const gate = createApprovalController( + { + mode: 'ask-all', + onRequest: (req) => { + requests.push(req) + return { approved: true, remember: true } + }, + }, + { sanitize: async (_n, input) => ({ ...(input as object), secret: '***' }) }, + ) + await inRun(events, async () => { + await gate.gate('db__write', { secret: 'p@ss' }, { readOnly: false }) + await gate.gate('db__write', { secret: 'p@ss' }, { readOnly: false }) + }) + assert.equal(requests.length, 1, 'the remembered tool is not asked again') + assert.deepEqual(requests[0].input, { secret: '***' }, 'the request shows the sanitized input') + assert.equal(requests[0].runId, 'run-1') + assert.equal(requests[0].step?.id, 's1') + assert.deepEqual( + events.map((e) => e.type), + ['tool.approval-requested', 'tool.approval-resolved'], + ) + assert.equal((events[1] as { automatic: boolean }).automatic, false) + assert.ok(gate.remembered.has('db__write')) +}) + +test('approval gate: denials throw ToolDeniedError; automatic ones say so', async () => { + const events: AgentEvent[] = [] + const noHandler = createApprovalController({ mode: 'ask-all' }) + await inRun(events, async () => { + await assert.rejects( + () => noHandler.gate('x', {}, { readOnly: true }), + (err: unknown) => + err instanceof ToolDeniedError && /no approval handler configured/.test(err.message), + ) + }) + assert.deepEqual(events.at(-1), { + type: 'tool.approval-resolved', + id: (events.at(-1) as { id: string }).id, + name: 'x', + approved: false, + reason: 'no approval handler configured', + automatic: true, + }) + + const readOnlyMode = createApprovalController({ mode: 'read-only' }) + await inRun(events, async () => { + await readOnlyMode.gate('reader', {}, { readOnly: true }) + await assert.rejects( + () => readOnlyMode.gate('writer', {}, { readOnly: false }), + ToolDeniedError, + ) + }) + + const userSaysNo = createApprovalController({ + mode: 'ask-writes', + onRequest: () => ({ approved: false, reason: 'too risky' }), + }) + await inRun(events, async () => { + await assert.rejects(() => userSaysNo.gate('w', {}, { readOnly: false }), /too risky/) + }) + const last = events.at(-1) as { automatic: boolean; approved: boolean } + assert.equal(last.approved, false) + assert.equal(last.automatic, false) +}) + +test('approval gate: a timeout denies; setMode switches at runtime', async () => { + const events: AgentEvent[] = [] + const gate = createApprovalController({ + mode: 'ask-all', + timeoutMs: 20, + onRequest: () => new Promise(() => {}), + }) + await inRun(events, async () => { + await assert.rejects( + () => gate.gate('slow', {}, { readOnly: false }), + /no decision within 20ms/, + ) + }) + gate.setMode('autopilot') + assert.equal(gate.getMode(), 'autopilot') + await gate.gate('slow', {}, { readOnly: false }) + assert.throws(() => gate.setMode('bogus' as never), /tool approval mode must be/) + assert.throws(() => createApprovalController({ mode: 'nope' as never }), /toolApproval.mode/) +}) + +test('approval gate: built-in tools never ask', async () => { + const gate = createApprovalController({ mode: 'ask-all', rules: { '*': 'deny' } }) + await gate.gate('load_skill', {}, { readOnly: true, builtIn: true }) +}) diff --git a/tests/compaction.test.ts b/tests/compaction.test.ts new file mode 100644 index 0000000..aaffcbb --- /dev/null +++ b/tests/compaction.test.ts @@ -0,0 +1,252 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { tool, type LanguageModel, type Tool } from 'ai' +import { MockLanguageModelV3 } from 'ai/test' +import { z } from 'zod' +import { + compactHistory, + estimateTokens, + isSummaryTurn, + resolveCompaction, + SUMMARY_PREFIX, + summarizeTrace, + truncateToolModelOutput, + withToolOutputLimit, +} from '../src/compaction.ts' +import { renderHistory, renderTrace, renderTraceSteps } from '../src/prompts.ts' +import type { IConversationTurn, IStepResult } from '../src/types.ts' + +const v3Usage = (input: number, output: number) => ({ + inputTokens: { total: input, noCache: input, cacheRead: 0, cacheWrite: 0 }, + outputTokens: { total: output, text: output, reasoning: 0 }, +}) + +// A generate-only mock that records the prompts it was given. +const summarizer = (text: string) => { + const prompts: string[] = [] + const model = new MockLanguageModelV3({ + doGenerate: async (options) => { + prompts.push(JSON.stringify(options.prompt)) + return { + content: [{ type: 'text', text }], + finishReason: { unified: 'stop', raw: undefined }, + usage: v3Usage(100, 20), + warnings: [], + } + }, + }) + return { model: model as LanguageModel, prompts } +} + +const failing = (): LanguageModel => + new MockLanguageModelV3({ + doGenerate: async () => { + throw new Error('provider down') + }, + }) + +const turns = (n: number, size = 40): IConversationTurn[] => + Array.from({ length: n }, (_, i) => ({ + role: i % 2 === 0 ? ('user' as const) : ('assistant' as const), + content: `turn ${i} ${'x'.repeat(size)}`, + })) + +test('estimateTokens is ceil(chars / 4)', () => { + assert.equal(estimateTokens(''), 0) + assert.equal(estimateTokens('abcd'), 1) + assert.equal(estimateTokens('abcde'), 2) +}) + +test('resolveCompaction defaults: auto on, 128k window, threshold at 50%', () => { + assert.deepEqual(resolveCompaction(undefined), { + auto: true, + contextWindowTokens: 128_000, + thresholdTokens: 64_000, + keepRecentTurns: 4, + keepRecentSteps: 3, + summaryMaxTokens: 1024, + maxToolOutputChars: 20_000, + clearToolResultsAfterTokens: 32_000, + keepToolResults: 3, + }) + const custom = resolveCompaction({ + auto: false, + contextWindowTokens: 8_000, + maxToolOutputChars: 0, + clearToolResultsAfterTokens: 0, + }) + assert.equal(custom.auto, false) + assert.equal(custom.thresholdTokens, 4_000) + assert.equal(custom.maxToolOutputChars, 0) + assert.equal(custom.clearToolResultsAfterTokens, 0) +}) + +test('compactHistory: below the threshold the input comes back untouched', async () => { + const { model, prompts } = summarizer('S') + const history = turns(10) + const r = await compactHistory(history, model, { thresholdTokens: 1_000_000 }) + assert.equal(r.compacted, false) + assert.equal(r.history, history) + assert.equal(prompts.length, 0, 'no model call below the threshold') +}) + +test('compactHistory: summarises all but the recent turns into one assistant summary turn', async () => { + const { model, prompts } = summarizer('They discussed invoices; total was 42.') + const history = turns(10) + const frozen = JSON.stringify(history) + const usages: number[] = [] + const r = await compactHistory(history, model, { + thresholdTokens: 10, + keepRecentTurns: 4, + onUsage: (u) => usages.push(u.totalTokens), + }) + assert.equal(r.compacted, true) + assert.equal(r.history.length, 5) + assert.equal(r.history[0].role, 'assistant') + assert.ok(r.history[0].content.startsWith(SUMMARY_PREFIX)) + assert.ok(isSummaryTurn(r.history[0])) + assert.deepEqual(r.history.slice(1), history.slice(6)) + assert.equal(r.summary, 'They discussed invoices; total was 42.') + assert.ok(r.afterTokens < r.beforeTokens) + assert.deepEqual(usages, [120]) + assert.equal(JSON.stringify(history), frozen, 'the input array is not mutated') + assert.ok(prompts[0].includes('turn 0') && !prompts[0].includes('turn 6')) + + // A second pass folds the earlier summary into the new one. + const again = await compactHistory([...r.history, ...turns(4)], model, { + force: true, + keepRecentTurns: 2, + }) + assert.equal(again.history.filter(isSummaryTurn).length, 1) +}) + +test('compactHistory: force bypasses the threshold; never throws on failure', async () => { + const { model } = summarizer('short') + const forced = await compactHistory(turns(6, 1), model, { force: true, keepRecentTurns: 2 }) + assert.equal(forced.compacted, true) + const history = turns(6) + const failed = await compactHistory(history, failing(), { force: true }) + assert.equal(failed.compacted, false) + assert.equal(failed.history, history) + // Nothing older than the kept window -> nothing to do. + const tiny = await compactHistory(turns(3), model, { force: true, keepRecentTurns: 4 }) + assert.equal(tiny.compacted, false) +}) + +test('renderHistory always keeps a leading summary turn', () => { + const history: IConversationTurn[] = [ + { role: 'assistant', content: `${SUMMARY_PREFIX}\n${'s'.repeat(3000)}` }, + ...turns(12), + ] + const out = renderHistory(history) + assert.ok(out.startsWith(`assistant: ${SUMMARY_PREFIX}`)) + assert.ok(out.includes('s'.repeat(3000)), 'the summary is not clipped at 1500 chars') + assert.match(out, /\(4 earlier turns omitted\)/) +}) + +const stepResult = (i: number): IStepResult => ({ + step: { id: `s${i}`, description: `do ${i}`, expectedOutcome: 'e' }, + summary: `result ${i}`, + toolCalls: [], + durationMs: 1, + blocked: false, +}) + +test('renderTrace with a summary view renders the summary + the recent steps verbatim', () => { + const trace = [stepResult(0), stepResult(1), stepResult(2)] + const out = renderTrace(trace, { summary: 'did 0 and 1', upTo: 2 }) + assert.equal( + out, + 'Summary of earlier steps (1-2): did 0 and 1\nStep 3: do 2\n Result: result 2\n Tool calls:\n (no tool calls)', + ) + assert.equal(renderTrace(trace, { upTo: 2 }), renderTrace(trace), 'no summary, no view') +}) + +test('summarizeTrace folds steps (and an earlier summary) into one summary; undefined on failure', async () => { + const { model, prompts } = summarizer('steps 1-2 found X') + const r = await summarizeTrace([stepResult(1)], 'earlier: found W', model, { + render: renderTraceSteps, + offset: 1, + }) + assert.equal(r?.summary, 'steps 1-2 found X') + assert.equal(r?.usage.totalTokens, 120) + assert.ok(prompts[0].includes('earlier: found W')) + assert.ok(prompts[0].includes('Step 2: do 1'), 'step numbers follow the offset') + assert.equal( + await summarizeTrace([stepResult(0)], undefined, failing(), { + render: renderTraceSteps, + offset: 0, + }), + undefined, + ) +}) + +test('truncateToolModelOutput clips text, serialised json and content text parts', () => { + assert.deepEqual(truncateToolModelOutput({ type: 'text', value: 'abcdefghij' }, 4), { + type: 'text', + value: 'abcd… [truncated 6 chars]', + }) + assert.deepEqual(truncateToolModelOutput({ type: 'text', value: 'abc' }, 4), { + type: 'text', + value: 'abc', + }) + const json = truncateToolModelOutput({ type: 'json', value: { rows: [1, 2, 3, 4, 5] } }, 10) + assert.equal(json.type, 'text') + assert.equal((json as { value: string }).value, '{"rows":[1… [truncated 10 chars]') + assert.deepEqual(truncateToolModelOutput({ type: 'json', value: [1] }, 10), { + type: 'json', + value: [1], + }) + const content = truncateToolModelOutput( + { type: 'content', value: [{ type: 'text', text: 'x'.repeat(20) }] }, + 5, + ) + assert.deepEqual(content, { + type: 'content', + value: [{ type: 'text', text: 'xxxxx… [truncated 15 chars]' }], + }) + // 0 = unlimited; errors pass through. + assert.deepEqual(truncateToolModelOutput({ type: 'text', value: 'abcdef' }, 0), { + type: 'text', + value: 'abcdef', + }) + assert.deepEqual(truncateToolModelOutput({ type: 'error-text', value: 'x'.repeat(50) }, 5), { + type: 'error-text', + value: 'x'.repeat(50), + }) +}) + +test('withToolOutputLimit: the model sees a clipped result, the raw output is untouched', async () => { + const big = tool({ + description: 'big', + inputSchema: z.object({}), + execute: async () => 'y'.repeat(100), + }) + const limited = withToolOutputLimit(big, 10) + const raw = await (limited.execute as (i: unknown, o: unknown) => Promise)({}, {}) + assert.equal(raw.length, 100) + const seen = await limited.toModelOutput!({ toolCallId: 'c', input: {}, output: raw }) + assert.deepEqual(seen, { type: 'text', value: 'yyyyyyyyyy… [truncated 90 chars]' }) + + // An existing toModelOutput runs first, then its value is clipped. + const custom = withToolOutputLimit( + tool({ + description: 'custom', + inputSchema: z.object({}), + execute: async () => ({ n: 1 }), + toModelOutput: () => ({ type: 'text', value: 'custom-rendering-that-is-long' }), + }), + 6, + ) + assert.deepEqual(await custom.toModelOutput!({ toolCallId: 'c', input: {}, output: { n: 1 } }), { + type: 'text', + value: 'custom… [truncated 23 chars]', + }) + // Objects become json (as the SDK would send them), clipped when long. + const obj = withToolOutputLimit(big as Tool, 1_000) + assert.deepEqual(await obj.toModelOutput!({ toolCallId: 'c', input: {}, output: { a: 1 } }), { + type: 'json', + value: { a: 1 }, + }) + assert.equal(withToolOutputLimit(big, 0), big, '0 = unlimited, no wrapper') +}) diff --git a/tests/config.test.ts b/tests/config.test.ts index d0e246e..68ec611 100644 --- a/tests/config.test.ts +++ b/tests/config.test.ts @@ -1,6 +1,6 @@ import assert from 'node:assert/strict' import test, { beforeEach } from 'node:test' -import { loadConfig } from '../src/cli/config.ts' +import { loadConfig, parseThinking, skillsDirFromEnv } from '../src/cli/config.ts' const CONFIG_KEYS = [ 'AGENT_PROVIDER_TYPE', @@ -30,6 +30,15 @@ const CONFIG_KEYS = [ 'AGENT_PROVIDER_OPTIONS', 'AGENT_PLANNER_PROVIDER_OPTIONS', 'AGENT_SYNTHESIZER_PROVIDER_OPTIONS', + 'AGENT_THINKING', + 'AGENT_MAX_INPUT_TOKENS', + 'AGENT_MAX_OUTPUT_TOKENS', + 'AGENT_MAX_REASONING_TOKENS', + 'AGENT_MAX_TOOL_CALLS', + 'AGENT_TOOL_APPROVAL', + 'AGENT_SKILLS_DIR', + 'AGENT_CONTEXT_WINDOW_TOKENS', + 'AGENT_COMPACTION', ] as const beforeEach(() => { @@ -229,3 +238,60 @@ test('loadConfig threads AGENT_PLANNER_PROVIDER_OPTIONS into the planner overrid providerOptions: { accountId: 'acc-from-env' }, }) }) + +test('loadConfig leaves every new knob undefined when its env var is unset', () => { + setMinimalOpenAi() + const c = loadConfig() + assert.equal(c.thinking, undefined) + assert.equal(c.limits, undefined) + assert.equal(c.maxToolCalls, undefined) + assert.equal(c.toolApproval, undefined) + assert.equal(c.compaction, undefined) + assert.equal(c.toolSelectionStrategy, undefined) + assert.equal(skillsDirFromEnv(), undefined) +}) + +test('AGENT_THINKING: levels, off, and a numeric budget', () => { + assert.equal(parseThinking(undefined), undefined) + assert.equal(parseThinking('HIGH'), 'high') + assert.equal(parseThinking('off'), 'none') + assert.equal(parseThinking('minimal'), 'minimal') + assert.deepEqual(parseThinking('4096'), { level: 'medium', budgetTokens: 4096 }) + assert.throws(() => parseThinking('turbo'), /AGENT_THINKING/) + setMinimalOpenAi() + process.env.AGENT_THINKING = 'xhigh' + assert.equal(loadConfig().thinking, 'xhigh') +}) + +test('loadConfig parses token limits, the tool-call cap and the approval mode', () => { + setMinimalOpenAi() + process.env.AGENT_MAX_INPUT_TOKENS = '1000' + process.env.AGENT_MAX_OUTPUT_TOKENS = '200' + process.env.AGENT_MAX_REASONING_TOKENS = '50' + process.env.AGENT_MAX_TOTAL_TOKENS = '1500' + process.env.AGENT_MAX_TOOL_CALLS = '12' + process.env.AGENT_TOOL_APPROVAL = 'ask-writes' + const c = loadConfig() + assert.deepEqual(c.limits, { maxInputTokens: 1000, maxOutputTokens: 200, maxReasoningTokens: 50 }) + assert.equal(c.maxTotalTokens, 1500, 'the legacy total cap keeps its slot') + assert.equal(c.maxToolCalls, 12) + assert.deepEqual(c.toolApproval, { mode: 'ask-writes' }) + process.env.AGENT_TOOL_APPROVAL = 'yolo' + assert.throws(() => loadConfig(), /AGENT_TOOL_APPROVAL/) +}) + +test('loadConfig parses compaction and the new tool strategies; AGENT_SKILLS_DIR is read raw', () => { + setMinimalOpenAi() + process.env.AGENT_COMPACTION = 'off' + process.env.AGENT_CONTEXT_WINDOW_TOKENS = '32000' + process.env.AGENT_TOOL_SELECTION_STRATEGY = 'search' + process.env.AGENT_SKILLS_DIR = ' ./skills ' + const c = loadConfig() + assert.deepEqual(c.compaction, { auto: false, contextWindowTokens: 32000 }) + assert.equal(c.toolSelectionStrategy, 'search') + assert.equal(skillsDirFromEnv(), './skills') + process.env.AGENT_TOOL_SELECTION_STRATEGY = 'auto' + assert.equal(loadConfig().toolSelectionStrategy, 'auto') + process.env.AGENT_COMPACTION = 'sometimes' + assert.throws(() => loadConfig(), /AGENT_COMPACTION/) +}) diff --git a/tests/dist-loadable.test.ts b/tests/dist-loadable.test.ts index a5b207a..ab23138 100644 --- a/tests/dist-loadable.test.ts +++ b/tests/dist-loadable.test.ts @@ -9,6 +9,7 @@ import test from 'node:test' // `npm test` works on a clean checkout. const HAS_CJS = existsSync('./dist/index.cjs') const HAS_ESM = existsSync('./dist/index.js') +const HAS_WORKER = existsSync('./dist/subagent-worker.js') test('dist/index.cjs is require()-able as CommonJS', { skip: !HAS_CJS }, () => { const out = execFileSync( @@ -38,3 +39,34 @@ test('dist/index.js is import()-able as ESM', { skip: !HAS_ESM }, () => { assert.ok(keys.includes('createAgent')) assert.ok(keys.includes('getCurrentRunId')) }) + +// createSubagentTool({ isolation: 'worker' }) resolves this entry next to +// the bundle; it must exist and load in a worker thread without crashing. +test('dist/subagent-worker.js loads in a worker thread', { skip: !HAS_ESM }, () => { + assert.ok(HAS_WORKER, 'dist/subagent-worker.js is built alongside dist/index.js') + const out = execFileSync( + 'node', + [ + '--input-type=module', + '-e', + `import { Worker } from 'node:worker_threads' + const w = new Worker(new URL('./dist/subagent-worker.js', 'file://' + process.cwd() + '/'), { execArgv: [] }) + w.on('error', (e) => { console.error(e); process.exit(1) }) + w.on('online', () => setTimeout(() => w.terminate().then(() => process.stdout.write('ok')), 200))`, + ], + { encoding: 'utf8' }, + ) + assert.equal(out, 'ok') +}) + +test('dist/index.cjs resolves the worker entry through import.meta.url', { skip: !HAS_CJS }, () => { + const out = execFileSync( + 'node', + [ + '-e', + "const m = require('./dist/index.cjs'); const t = m.createSubagentTool({ name: 'x', description: 'x', config: { clientName: 'c', providerType: 'openai', apiKey: 'k', model: 'm', mcpServers: {}, maxIterations: 1, maxStepsPerTask: 1, logLevel: 'none' } }); process.stdout.write(typeof t.execute)", + ], + { encoding: 'utf8' }, + ) + assert.equal(out, 'function') +}) diff --git a/tests/features-integration.test.ts b/tests/features-integration.test.ts new file mode 100644 index 0000000..1a13b25 --- /dev/null +++ b/tests/features-integration.test.ts @@ -0,0 +1,760 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { isMainThread } from 'node:worker_threads' +import { tool, type ToolSet } from 'ai' +import { z } from 'zod' +import { + createAgent, + createSubagentTool, + markReadOnly, + type AgentEvent, + type IAgent, + type IAgentConfig, + type IRunSnapshot, + type IToolApprovalRequest, +} from '../src/index.ts' +import { startLocalMcp } from './helpers/mcp-server.ts' +import { + hasToolResult, + startLocalOpenAI, + systemOf, + toolNames, + toolResults, + userOf, + type IChatRequest, + type ILocalOpenAI, + type IScriptedReply, + type Script, +} from './helpers/openai-server.ts' + +// Every feature of the loop end-to-end over real sockets: a scripted +// OpenAI-compatible endpoint (the real provider package, the real wire +// format) and, where tools come from MCP, a real MCP server. The script +// answers per stage, recognised by the system prompt. + +type Stage = + 'planner' | 'executor' | 'replanner' | 'synthesizer' | 'compact-history' | 'compact-trace' + +const stageOf = (req: IChatRequest): Stage => { + const system = systemOf(req) + if (system.includes('You are the Planner')) return 'planner' + if (system.includes('You are the Replanner')) return 'replanner' + if (system.includes('You are the Synthesizer')) return 'synthesizer' + if (system.includes('You compress conversation history')) return 'compact-history' + if (system.includes('You compress the execution trace')) return 'compact-trace' + return 'executor' +} + +// Which plan step an executor request is for. +const stepOf = (req: IChatRequest): string => + /CURRENT STEP to execute \(id=([^)]+)\)/.exec(userOf(req))?.[1] ?? '?' + +const planJson = (steps: { id: string; suggestedTools?: string[] }[], skills?: string[]) => + JSON.stringify({ + thought: 'Do it step by step.', + steps: steps.map((s) => ({ + id: s.id, + description: `Step ${s.id}`, + expectedOutcome: `${s.id} done`, + suggestedTools: s.suggestedTools ?? [], + })), + ...(skills ? { skills } : {}), + }) + +const config = (baseURL: string, extra: Partial = {}): IAgentConfig => ({ + clientName: 'features-test', + providerType: 'openai-compatible', + baseURL, + apiKey: 'not-a-real-key', + model: 'scripted', + mcpServers: {}, + maxIterations: 6, + maxStepsPerTask: 6, + logLevel: 'none', + ...extra, +}) + +const withLlm = async ( + script: Script, + body: (ctx: { llm: ILocalOpenAI; open: (c: IAgentConfig) => Promise }) => Promise, +): Promise => { + const llm = await startLocalOpenAI(script) + const agents: IAgent[] = [] + try { + await body({ + llm, + open: async (c) => { + const agent = await createAgent(c) + agents.push(agent) + return agent + }, + }) + } finally { + for (const agent of agents) await agent.close().catch(() => {}) + await llm.close() + } +} + +const collect = () => { + const events: AgentEvent[] = [] + return { events, onEvent: (e: AgentEvent) => events.push(e) } +} + +const ofType = (events: AgentEvent[], type: T) => + events.filter((e): e is Extract => e.type === type) + +// ── tool approval ────────────────────────────────────────────────────────── + +test('approval: ask-writes asks for a write tool once (remember), read-only tools run unasked', async () => { + const mcp = await startLocalMcp() + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') { + return { text: planJson([{ id: 's1', suggestedTools: ['files__echo', 'files__secret'] }]) } + } + if (stage === 'synthesizer') return { text: 'done' } + const results = toolResults(req).length + if (results === 0) { + return { + toolCalls: [ + { name: 'files__secret', args: {} }, + { name: 'files__echo', args: { text: 'a' } }, + ], + } + } + if (results === 2) return { toolCalls: [{ name: 'files__echo', args: { text: 'b' } }] } + return { text: 'Echoed twice and read the secret.' } + } + try { + await withLlm(script, async ({ llm, open }) => { + const requests: IToolApprovalRequest[] = [] + const agent = await open( + config(llm.baseURL, { + mcpServers: { files: { url: mcp.url } }, + toolApproval: { + mode: 'ask-writes', + onRequest: (req) => { + requests.push(req) + return { approved: true, remember: true } + }, + }, + }), + ) + assert.equal(agent.listTools().find((t) => t.name === 'files__secret')?.readOnly, true) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Read the secret and echo twice', onEvent }) + + assert.deepEqual( + requests.map((r) => r.toolName), + ['files__echo'], + 'asked once, for the write tool only', + ) + assert.deepEqual(requests[0].input, { text: 'a' }) + assert.equal(requests[0].readOnly, false) + assert.equal(requests[0].step?.id, 's1') + // secret and the first echo ran in parallel: compare as a multiset. + assert.deepEqual(mcp.calls.map((c) => c.name).sort(), ['echo', 'echo', 'secret']) + assert.ok(result.trace[0].toolCalls.every((c) => c.ok)) + assert.equal(ofType(events, 'tool.approval-requested').length, 1) + const resolved = ofType(events, 'tool.approval-resolved') + assert.equal(resolved.length, 1) + assert.equal(resolved[0].approved, true) + assert.equal(resolved[0].automatic, false) + assert.ok(resolved[0].runId, 'gate events are tagged with the run') + + // The executor's system prompt is run-stable across its LLM steps. + const execSystems = new Set( + llm.requests.filter((r) => stageOf(r) === 'executor').map(systemOf), + ) + assert.equal(execSystems.size, 1) + }) + } finally { + await mcp.close() + } +}) + +test('approval: a denied call fails with ToolDeniedError; read-only mode denies automatically', async () => { + const mcp = await startLocalMcp() + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') return { text: planJson([{ id: 's1' }]) } + if (stage === 'synthesizer') return { text: 'could not echo' } + if (hasToolResult(req)) return { text: 'The echo was denied.' } + return { toolCalls: [{ name: 'files__echo', args: { text: 'nope' } }] } + } + try { + await withLlm(script, async ({ llm, open }) => { + const agent = await open( + config(llm.baseURL, { + mcpServers: { files: { url: mcp.url } }, + toolApproval: { mode: 'ask-all', onRequest: () => ({ approved: false, reason: 'no' }) }, + }), + ) + const first = await agent.run({ input: 'Echo nope' }) + const call = first.trace[0].toolCalls[0] + assert.equal(call.ok, false) + assert.match(String(JSON.parse(JSON.stringify(call.output))), /denied by the user: no/) + // The model saw the denial as the tool result. + const denialSeen = llm.requests.some((r) => + toolResults(r).some((t) => t.includes('Tool call denied by the user')), + ) + assert.ok(denialSeen) + assert.deepEqual(mcp.calls, [], 'the denied call never reached the server') + + // The runtime switch applies to the next call: read-only denies writes + // without asking. + agent.setToolApprovalMode('read-only') + assert.equal(agent.getToolApprovalMode(), 'read-only') + const { events, onEvent } = collect() + await agent.run({ input: 'Echo nope', onEvent }) + assert.equal(ofType(events, 'tool.approval-requested').length, 0) + const resolved = ofType(events, 'tool.approval-resolved') + assert.equal(resolved.length, 1) + assert.equal(resolved[0].automatic, true) + assert.equal(resolved[0].approved, false) + assert.deepEqual(mcp.calls, []) + }) + } finally { + await mcp.close() + } +}) + +// ── tool search ──────────────────────────────────────────────────────────── + +test('tool search: >40 tools switch auto to search; find_tools activates a tool for the next step and later steps', async () => { + const weatherCalls: unknown[] = [] + const tools: ToolSet = {} + for (let i = 0; i < 45; i++) { + tools[`bulk_${i}`] = tool({ + description: `Bulk utility number ${i}`, + inputSchema: z.object({}), + execute: async () => i, + }) + } + tools.weather_lookup = markReadOnly( + tool({ + description: 'Look up the weather forecast for a city', + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }) => { + weatherCalls.push(city) + return `Sunny in ${city}` + }, + }), + ) + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') return { text: planJson([{ id: 's1' }, { id: 's2' }]) } + if (stage === 'synthesizer') return { text: 'It is sunny in Oslo.' } + if (stepOf(req) === 's2') return { text: 'Nothing else to do.' } + const results = toolResults(req).length + if (results === 0) + return { toolCalls: [{ name: 'find_tools', args: { query: 'weather forecast' } }] } + if (results === 1) return { toolCalls: [{ name: 'weather_lookup', args: { city: 'Oslo' } }] } + return { text: 'Sunny in Oslo.' } + } + await withLlm(script, async ({ llm, open }) => { + const agent = await open(config(llm.baseURL, { tools })) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Weather in Oslo?', onEvent }) + + assert.equal(result.text, 'It is sunny in Oslo.') + assert.deepEqual(weatherCalls, ['Oslo']) + const exec = llm.requests.filter((r) => stageOf(r) === 'executor') + assert.deepEqual(toolNames(exec[0]), ['find_tools'], 'a step starts with the built-ins only') + assert.deepEqual(toolNames(exec[1]).sort(), ['find_tools', 'weather_lookup']) + assert.ok(!toolNames(exec[1]).includes('bulk_0')) + const s2 = exec.find((r) => stepOf(r) === 's2')! + assert.ok(toolNames(s2).includes('weather_lookup'), 'discovered tools carry into later steps') + const discovered = ofType(events, 'tools.discovered') + assert.equal(discovered.length, 1) + assert.equal(discovered[0].query, 'weather forecast') + assert.equal(discovered[0].names[0], 'weather_lookup') + // The planner sees the abridged catalogue and the search-mode rule. + const planner = llm.requests.find((r) => stageOf(r) === 'planner')! + assert.match(systemOf(planner), /tool-search mode/) + assert.match(systemOf(planner), /\[\]\n- bulk_0: Bulk utility number 0/) + assert.equal(new Set(exec.map(systemOf)).size, 1, 'one executor system prompt per run') + }) +}) + +// ── token limits, tool-call cap, thinking, usage details ─────────────────── + +test('limits: a token cap stops a runaway tool loop at the next LLM step and skips to synthesis', async () => { + let noopCalls = 0 + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') return { text: planJson([{ id: 's1' }, { id: 's2' }, { id: 's3' }]) } + if (stage === 'synthesizer') return { text: 'Partial answer.' } + return { toolCalls: [{ name: 'noop', args: {} }] } // never stops on its own + } + await withLlm(script, async ({ llm, open }) => { + const agent = await open( + config(llm.baseURL, { + maxStepsPerTask: 10, + limits: { maxTotalTokens: 40, perCall: { synthesizer: 77 } }, + tools: { + noop: tool({ + description: 'Does nothing', + inputSchema: z.object({}), + execute: async () => { + noopCalls++ + return 'ok' + }, + }), + }, + }), + ) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Loop forever', onEvent }) + + // 15 tokens per call: plan 15, then the executor's 2nd LLM step reaches + // 15 + 30 = 45 >= 40 and stops the loop. + assert.equal(llm.requests.filter((r) => stageOf(r) === 'executor').length, 2) + assert.equal(noopCalls, 2) + assert.equal(result.trace.length, 1, 'no further plan step ran') + assert.match(result.trace[0].summary, /token budget/) + assert.deepEqual( + ofType(events, 'budget.exceeded').map(({ kind, tokens, cap }) => ({ kind, tokens, cap })), + [{ kind: 'total', tokens: 45, cap: 40 }], + ) + assert.equal(result.text, 'Partial answer.', 'synthesis still runs') + assert.equal(result.usage.totalTokens, 60) + assert.equal(ofType(events, 'replan.decision').length, 0, 'no replanner after the breach') + const synth = llm.requests.find((r) => stageOf(r) === 'synthesizer') as unknown as { + max_tokens?: number + } + assert.equal(synth.max_tokens, 77, 'perCall.synthesizer caps the synthesis call') + }) +}) + +test('limits: maxToolCalls fails calls past the cap and reports kind "tool-calls"', async () => { + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') return { text: planJson([{ id: 's1' }, { id: 's2' }]) } + if (stage === 'synthesizer') return { text: 'Stopped.' } + return { + toolCalls: [ + { name: 'noop', args: {} }, + { name: 'noop', args: {} }, + { name: 'noop', args: {} }, + ], + } + } + await withLlm(script, async ({ llm, open }) => { + let calls = 0 + const agent = await open( + config(llm.baseURL, { + maxToolCalls: 2, + tools: { + noop: tool({ + description: 'Does nothing', + inputSchema: z.object({}), + execute: async () => ++calls, + }), + }, + }), + ) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Call a lot', onEvent }) + assert.equal(calls, 2) + // The three calls ran in parallel; results arrive in completion order. + const outcomes = result.trace[0].toolCalls + assert.equal(outcomes.filter((c) => c.ok).length, 2) + const failed = outcomes.filter((c) => !c.ok) + assert.equal(failed.length, 1) + assert.match(String(JSON.parse(JSON.stringify(failed[0].output))), /tool-call budget exhausted/) + assert.deepEqual( + ofType(events, 'budget.exceeded').map(({ kind, tokens, cap }) => ({ kind, tokens, cap })), + [{ kind: 'tool-calls', tokens: 2, cap: 2 }], + ) + assert.equal(result.trace.length, 1) + assert.equal(llm.requests.filter((r) => stageOf(r) === 'executor').length, 1) + }) +}) + +test('thinking: reasoning reaches the wire, thoughts stream, usage details accumulate', async () => { + const usage = { prompt: 100, completion: 20, cached: 60, reasoning: 8 } + const script: Script = (req): IScriptedReply => { + const stage = stageOf(req) + if (stage === 'planner') return { text: planJson([{ id: 's1' }]), usage } + if (stage === 'synthesizer') return { text: 'Answer.', reasoning: 'Summing up.', usage } + return { text: 'Did it.', reasoning: 'Let me think.', usage } + } + await withLlm(script, async ({ llm, open }) => { + const agent = await open( + config(llm.baseURL, { thinking: 'high', stageThinking: { planner: false } }), + ) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Think about it', onEvent }) + + const effort = (stage: Stage) => + (llm.requests.find((r) => stageOf(r) === stage) as unknown as { reasoning_effort?: string }) + .reasoning_effort + assert.equal(effort('executor'), 'high') + assert.equal(effort('synthesizer'), 'high') + assert.equal(effort('planner'), undefined, 'stageThinking.planner: false wins') + assert.equal( + ofType(events, 'step.reasoning-delta') + .map((e) => e.delta) + .join(''), + 'Let me think.', + ) + assert.equal( + ofType(events, 'final.reasoning-delta') + .map((e) => e.delta) + .join(''), + 'Summing up.', + ) + assert.equal(result.text, 'Answer.') + assert.deepEqual(result.usage, { + inputTokens: 300, + outputTokens: 60, + totalTokens: 360, + reasoningTokens: 24, + cachedInputTokens: 180, + cacheWriteTokens: 0, + }) + }) +}) + +// ── skills ───────────────────────────────────────────────────────────────── + +test('skills: the plan activates one, load_skill another; instructions reach later stages', async () => { + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') { + return { text: planJson([{ id: 's1' }, { id: 's2' }], ['alpha', 'ghost']) } + } + if (stage === 'synthesizer') return { text: 'Followed both skills.' } + if (stepOf(req) === 's2') return { text: 's2 done' } + if (hasToolResult(req)) return { text: 'Loaded beta.' } + return { toolCalls: [{ name: 'load_skill', args: { name: 'beta' } }] } + } + await withLlm(script, async ({ llm, open }) => { + const logs: string[] = [] + const agent = await createAgent( + config(llm.baseURL, { + logLevel: 'warn', + // Built-ins never need approval, even in ask-all without a handler. + toolApproval: { mode: 'ask-all' }, + skills: [ + { name: 'alpha', description: 'Alpha work', content: 'ALPHA-INSTRUCTIONS' }, + { + name: 'beta', + description: 'Beta work', + content: 'BETA-INSTRUCTIONS', + files: [{ path: 'ref.md', content: 'ref' }], + }, + ], + }), + (e) => { + if (e.type === 'log') logs.push(e.message) + }, + ) + try { + assert.deepEqual(agent.listSkills(), [ + { name: 'alpha', description: 'Alpha work' }, + { name: 'beta', description: 'Beta work' }, + ]) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Do alpha and beta work', onEvent }) + + assert.deepEqual(result.plan.skills, ['alpha']) + assert.ok(logs.some((m) => m.includes('dropped unknown skills: ghost'))) + assert.deepEqual( + ofType(events, 'skill.activated').map((e) => `${e.name}:${e.by}`), + ['alpha:plan', 'beta:tool'], + ) + const call = result.trace[0].toolCalls[0] + assert.equal(call.ok, true) + assert.deepEqual(call.output, { + name: 'beta', + content: 'BETA-INSTRUCTIONS', + files: ['ref.md'], + }) + + const planner = llm.requests.find((r) => stageOf(r) === 'planner')! + assert.match(systemOf(planner), /SKILLS:\n- alpha: Alpha work\n- beta: Beta work/) + const exec = llm.requests.filter((r) => stageOf(r) === 'executor') + const s1 = systemOf(exec.find((r) => stepOf(r) === 's1')!) + const s2 = systemOf(exec.find((r) => stepOf(r) === 's2')!) + assert.ok(s1.includes('ALPHA-INSTRUCTIONS') && !s1.includes('BETA-INSTRUCTIONS')) + assert.ok(s2.includes('ALPHA-INSTRUCTIONS') && s2.includes('BETA-INSTRUCTIONS')) + assert.ok(toolNames(exec[0]).includes('load_skill')) + const synth = systemOf(llm.requests.find((r) => stageOf(r) === 'synthesizer')!) + assert.ok(synth.includes('ALPHA-INSTRUCTIONS') && synth.includes('BETA-INSTRUCTIONS')) + } finally { + await agent.close() + } + }) +}) + +test('skills: a host tool named load_skill collides with the built-in', async () => { + await assert.rejects( + () => + createAgent( + config('http://127.0.0.1:9/v1', { + skills: [{ name: 'a', description: 'd', content: 'c' }], + tools: { + load_skill: tool({ + description: 'x', + inputSchema: z.object({}), + execute: async () => 1, + }), + }, + }), + ), + /collides with a built-in skill tool/, + ) +}) + +// ── compaction ───────────────────────────────────────────────────────────── + +test('compaction: history at run start and the trace before later stages, with a low threshold', async () => { + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') return { text: planJson([{ id: 's1' }, { id: 's2' }, { id: 's3' }]) } + if (stage === 'compact-history') return { text: 'HISTORY-SUMMARY' } + if (stage === 'compact-trace') return { text: 'TRACE-SUMMARY' } + if (stage === 'synthesizer') return { text: 'All three done.' } + return { text: `${stepOf(req)} result` } + } + await withLlm(script, async ({ llm, open }) => { + let finalSnapshot: IRunSnapshot | undefined + const agent = await open( + config(llm.baseURL, { + compaction: { thresholdTokens: 1, keepRecentSteps: 1, keepRecentTurns: 2 }, + persistence: { + onRunComplete: (s) => { + finalSnapshot = s + }, + }, + }), + ) + const history = Array.from({ length: 6 }, (_, i) => ({ + role: i % 2 ? ('assistant' as const) : ('user' as const), + content: `old turn ${i}`, + })) + const frozen = JSON.stringify(history) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Do three things', history, onEvent }) + + assert.equal(JSON.stringify(history), frozen, "the caller's history is not mutated") + assert.equal(result.compactedHistory?.length, 3) + assert.match( + result.compactedHistory![0].content, + /^\[Summary of earlier conversation\]\nHISTORY-SUMMARY/, + ) + assert.deepEqual(result.compactedHistory!.slice(1), history.slice(4)) + const planner = llm.requests.find((r) => stageOf(r) === 'planner')! + assert.ok(userOf(planner).includes('HISTORY-SUMMARY')) + assert.ok(!userOf(planner).includes('old turn 0')) + + const compacted = ofType(events, 'context.compacted') + assert.deepEqual( + compacted.map((e) => e.scope), + ['history', 'trace', 'trace'], + ) + // Before step 3: steps [0,1) folded; before synthesis: [0,2). + const s3 = llm.requests.find((r) => stageOf(r) === 'executor' && stepOf(r) === 's3')! + assert.match(userOf(s3), /Summary of earlier steps \(1-1\): TRACE-SUMMARY/) + assert.match(userOf(s3), /Step 2: Step s2/) + const synth = llm.requests.find((r) => stageOf(r) === 'synthesizer')! + assert.match(userOf(synth), /Summary of earlier steps \(1-2\): TRACE-SUMMARY\nStep 3: Step s3/) + // Compaction calls never think and their usage counts (phase 'compact'). + assert.equal(ofType(events, 'usage').filter((e) => e.phase === 'compact').length, 3) + assert.equal(result.trace.length, 3, 'the trace itself stays intact') + assert.equal(finalSnapshot?.traceSummary, 'TRACE-SUMMARY') + assert.equal(finalSnapshot?.traceSummaryUpTo, 2) + + // Manual compaction of a history. + const manual = await agent.compact({ history, force: true }) + assert.equal(manual.compacted, true) + assert.equal(manual.history.length, 3) + assert.equal(manual.summary, 'HISTORY-SUMMARY') + }) +}) + +// ── subagents ────────────────────────────────────────────────────────────── + +const subagentScript: Script = (req) => { + const stage = stageOf(req) + const child = systemOf(req).includes('CHILD AGENT') + if (stage === 'planner') return { text: planJson([{ id: 's1' }]) } + if (child) { + if (stage === 'synthesizer') return { text: 'child answer: host-value-42' } + if (hasToolResult(req)) return { text: `child saw ${toolResults(req)[0]}` } + return { toolCalls: [{ name: 'host_lookup', args: { key: 'k' } }] } + } + if (stage === 'synthesizer') return { text: 'Parent done.' } + if (hasToolResult(req)) return { text: `parent got ${toolResults(req)[0]}` } + return { + toolCalls: [{ name: 'researcher', args: { task: 'Find the value for key k' } }], + } +} + +for (const isolation of ['worker', 'in-process'] as const) { + test(`subagent (${isolation}): proxies a host tool, returns its text, folds usage into the parent`, async () => { + await withLlm(subagentScript, async ({ llm, open }) => { + const hostCalls: { key: string; mainThread: boolean }[] = [] + const host_lookup = tool({ + description: 'Look up a value on the host', + inputSchema: z.object({ key: z.string() }), + execute: async ({ key }) => { + hostCalls.push({ key, mainThread: isMainThread }) + return 'host-value-42' + }, + }) + const researcher = createSubagentTool({ + name: 'researcher', + description: 'Delegate a research task', + isolation, + config: config(llm.baseURL, { systemPrompt: 'CHILD AGENT', maxIterations: 2 }), + tools: { host_lookup }, + timeoutMs: 20_000, + }) + const agent = await open(config(llm.baseURL, { tools: { researcher } })) + const { events, onEvent } = collect() + const result = await agent.run({ input: 'Research it', onEvent }) + + assert.deepEqual(hostCalls, [{ key: 'k', mainThread: true }]) + const call = result.trace[0].toolCalls[0] + assert.equal(call.name, 'researcher') + assert.equal(call.ok, true, JSON.stringify(call.output)) + assert.equal(call.output, 'child answer: host-value-42') + assert.equal(result.text, 'Parent done.') + + const start = ofType(events, 'subagent.start') + const complete = ofType(events, 'subagent.complete') + assert.equal(start.length, 1) + assert.equal(start[0].task, 'Find the value for key k') + assert.equal(complete.length, 1) + assert.equal(complete[0].id, start[0].id) + assert.equal(complete[0].text, 'child answer: host-value-42') + // Child: plan + 2 executor steps + synthesis = 4 calls x 15 tokens. + assert.equal(complete[0].usage.totalTokens, 60) + const forwarded = ofType(events, 'subagent.event').map((e) => e.event.type) + assert.ok(forwarded.includes('plan.created') && forwarded.includes('final')) + const childUsage = ofType(events, 'usage').filter((e) => e.phase === 'subagent') + assert.equal( + childUsage.reduce((n, e) => n + e.usage.totalTokens, 0), + 60, + ) + // Parent: 4 calls x 15 + the child's 60. + assert.equal(result.usage.totalTokens, 120) + }) + }) +} + +test('subagent: worker isolation refuses a config with functions', () => { + assert.throws( + () => + createSubagentTool({ + name: 'bad', + description: 'x', + config: config('http://127.0.0.1:9/v1', { inputSanitizer: (_n, i) => i }), + }), + /config\.inputSanitizer is a function/, + ) + assert.throws( + () => + createSubagentTool({ + name: 'bad', + description: 'x', + config: config('http://127.0.0.1:9/v1', { + tools: { + t: tool({ description: 'x', inputSchema: z.object({}), execute: async () => 1 }), + }, + }), + }), + /use the `tools` option/, + ) +}) + +test('maxPlanSteps caps the plan and is stated in the planner prompt', async () => { + const script: Script = (req) => { + const stage = stageOf(req) + if (stage === 'planner') { + return { text: planJson([{ id: 'a' }, { id: 'b' }, { id: 'c' }, { id: 'd' }]) } + } + if (stage === 'synthesizer') return { text: 'ok' } + return { text: `${stepOf(req)} done` } + } + await withLlm(script, async ({ llm, open }) => { + const agent = await open(config(llm.baseURL, { maxPlanSteps: 2 })) + const result = await agent.run({ input: 'Many steps' }) + assert.deepEqual( + result.plan.steps.map((s) => s.id), + ['a', 'b'], + ) + assert.equal(result.trace.length, 2) + const planner = llm.requests.find((r) => stageOf(r) === 'planner')! + assert.match(systemOf(planner), /hard cap is 2\)/) + }) +}) + +// A child whose only tool never returns: the parent's timeout / abort must +// still end the call (and the worker). +const hangingChildScript: Script = (req) => { + const stage = stageOf(req) + const child = systemOf(req).includes('CHILD AGENT') + if (stage === 'planner') return { text: planJson([{ id: 's1' }]) } + if (stage === 'synthesizer') return { text: child ? 'never' : 'Parent gave up waiting.' } + if (child) return { toolCalls: [{ name: 'hang', args: {} }] } + if (hasToolResult(req)) return { text: 'The subagent failed.' } + return { toolCalls: [{ name: 'researcher', args: { task: 'Wait forever' } }] } +} + +const hangTool = () => + tool({ + description: 'Never returns', + inputSchema: z.object({}), + execute: () => new Promise(() => {}), + }) + +test('subagent (worker): timeoutMs fails the call; the parent run carries on', async () => { + await withLlm(hangingChildScript, async ({ llm, open }) => { + const researcher = createSubagentTool({ + name: 'researcher', + description: 'Delegate', + config: config(llm.baseURL, { systemPrompt: 'CHILD AGENT' }), + tools: { hang: hangTool() }, + timeoutMs: 300, + }) + const agent = await open(config(llm.baseURL, { tools: { researcher } })) + const { events, onEvent } = collect() + const started = Date.now() + const result = await agent.run({ input: 'Research', onEvent }) + assert.ok(Date.now() - started < 10_000) + const call = result.trace[0].toolCalls[0] + assert.equal(call.ok, false) + assert.match(String((call.output as Error).message), /timed out after 300ms/) + assert.equal(ofType(events, 'subagent.error').length, 1) + assert.equal(result.text, 'Parent gave up waiting.') + }) +}) + +test('subagent (worker): aborting the parent run aborts the child', async () => { + await withLlm(hangingChildScript, async ({ llm, open }) => { + const researcher = createSubagentTool({ + name: 'researcher', + description: 'Delegate', + config: config(llm.baseURL, { systemPrompt: 'CHILD AGENT' }), + tools: { hang: hangTool() }, + }) + const agent = await open(config(llm.baseURL, { tools: { researcher } })) + const ac = new AbortController() + const onEvent = (e: AgentEvent) => { + // Abort once the child is stuck inside its tool call. + if (e.type === 'subagent.event' && e.event.type === 'step.tool-call') { + setTimeout(() => ac.abort(), 20) + } + } + const started = Date.now() + await assert.rejects( + () => agent.run({ input: 'Research', signal: ac.signal, onEvent }), + (err: unknown) => (err as Error).name === 'AbortError', + ) + assert.ok(Date.now() - started < 10_000) + }) +}) diff --git a/tests/helpers/mcp-server.ts b/tests/helpers/mcp-server.ts index 4bb148d..9d7adc3 100644 --- a/tests/helpers/mcp-server.ts +++ b/tests/helpers/mcp-server.ts @@ -147,6 +147,7 @@ export const startLocalMcp = async (opts: ILocalMcpOptions = {}): Promise { calls.push({ name: 'secret', args: {} }) diff --git a/tests/helpers/openai-server.ts b/tests/helpers/openai-server.ts index eded201..054bf65 100644 --- a/tests/helpers/openai-server.ts +++ b/tests/helpers/openai-server.ts @@ -24,6 +24,10 @@ export interface IScriptedReply { text?: string /** Tool calls to emit instead of text. */ toolCalls?: { name: string; args: unknown }[] + /** Streamed first as `reasoning_content` deltas (the model's thoughts). */ + reasoning?: string + /** Usage to report instead of the default 10 in / 5 out. */ + usage?: { prompt?: number; completion?: number; cached?: number; reasoning?: number } } export type Script = (request: IChatRequest) => IScriptedReply @@ -42,13 +46,38 @@ const CHUNK_HEAD = { model: 'scripted', } -const USAGE = { prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 } +const usageOf = (reply: IScriptedReply) => { + const prompt = reply.usage?.prompt ?? 10 + const completion = reply.usage?.completion ?? 5 + return { + prompt_tokens: prompt, + completion_tokens: completion, + total_tokens: prompt + completion, + ...(reply.usage?.cached !== undefined + ? { prompt_tokens_details: { cached_tokens: reply.usage.cached } } + : {}), + ...(reply.usage?.reasoning !== undefined + ? { completion_tokens_details: { reasoning_tokens: reply.usage.reasoning } } + : {}), + } +} const sse = (payload: unknown): string => `data: ${JSON.stringify(payload)}\n\n` const streamBody = (reply: IScriptedReply): string => { const parts: string[] = [] parts.push(sse({ ...CHUNK_HEAD, choices: [{ index: 0, delta: { role: 'assistant' } }] })) + if (reply.reasoning) { + const size = Math.max(1, Math.ceil(reply.reasoning.length / 2)) + for (let i = 0; i < reply.reasoning.length; i += size) { + parts.push( + sse({ + ...CHUNK_HEAD, + choices: [{ index: 0, delta: { reasoning_content: reply.reasoning.slice(i, i + size) } }], + }), + ) + } + } if (reply.toolCalls?.length) { reply.toolCalls.forEach((call, index) => { @@ -106,7 +135,7 @@ const streamBody = (reply: IScriptedReply): string => { parts.push(sse({ ...CHUNK_HEAD, choices: [{ index: 0, delta: {}, finish_reason: 'stop' }] })) } - parts.push(sse({ ...CHUNK_HEAD, choices: [], usage: USAGE })) + parts.push(sse({ ...CHUNK_HEAD, choices: [], usage: usageOf(reply) })) parts.push('data: [DONE]\n\n') return parts.join('') } @@ -134,7 +163,7 @@ const jsonBody = (reply: IScriptedReply): string => finish_reason: reply.toolCalls?.length ? 'tool_calls' : 'stop', }, ], - usage: USAGE, + usage: usageOf(reply), }) /** Start the endpoint on an ephemeral port. `script` decides each answer. */ @@ -187,3 +216,25 @@ export const wantsSchema = (req: IChatRequest, name: string): boolean => /** True when the conversation already carries a tool result. */ export const hasToolResult = (req: IChatRequest): boolean => req.messages.some((m) => m.role === 'tool' || m.tool_call_id !== undefined) + +/** The system prompt of a request ('' when none). */ +export const systemOf = (req: IChatRequest): string => { + const system = req.messages.find((m) => m.role === 'system')?.content + return typeof system === 'string' ? system : JSON.stringify(system ?? '') +} + +/** The (first) user message of a request ('' when none). */ +export const userOf = (req: IChatRequest): string => { + const user = req.messages.find((m) => m.role === 'user')?.content + return typeof user === 'string' ? user : JSON.stringify(user ?? '') +} + +/** Tool results already in the conversation, oldest first. */ +export const toolResults = (req: IChatRequest): string[] => + req.messages + .filter((m) => m.role === 'tool') + .map((m) => (typeof m.content === 'string' ? m.content : JSON.stringify(m.content))) + +/** Names of the tools offered in a request. */ +export const toolNames = (req: IChatRequest): string[] => + (req.tools ?? []).map((t) => t.function?.name ?? '').filter(Boolean) diff --git a/tests/limits.test.ts b/tests/limits.test.ts new file mode 100644 index 0000000..5d48acc --- /dev/null +++ b/tests/limits.test.ts @@ -0,0 +1,105 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { checkLimits, limitsStopCondition, resolveLimits, sumStepUsage } from '../src/limits.ts' +import { addUsage, normalizeUsage } from '../src/utils.ts' + +const usage = (input: number, output: number, reasoning = 0) => ({ + inputTokens: input, + outputTokens: output, + totalTokens: input + output, + reasoningTokens: reasoning, +}) + +test('checkLimits: no limits, no breach', () => { + assert.equal(checkLimits(usage(1e9, 1e9), undefined), undefined) + assert.equal(checkLimits(usage(1e9, 1e9), {}), undefined) +}) + +test('checkLimits reports the first cap reached (>=) in input/output/reasoning/total order', () => { + assert.deepEqual(checkLimits(usage(100, 5), { maxInputTokens: 100 }), { + kind: 'input', + tokens: 100, + cap: 100, + }) + assert.equal(checkLimits(usage(99, 5), { maxInputTokens: 100 }), undefined) + assert.deepEqual(checkLimits(usage(1, 50), { maxOutputTokens: 40 }), { + kind: 'output', + tokens: 50, + cap: 40, + }) + assert.deepEqual(checkLimits(usage(1, 50, 30), { maxReasoningTokens: 30 }), { + kind: 'reasoning', + tokens: 30, + cap: 30, + }) + assert.deepEqual(checkLimits(usage(10, 10), { maxTotalTokens: 20, maxInputTokens: 1_000 }), { + kind: 'total', + tokens: 20, + cap: 20, + }) + // Input is checked before total. + assert.equal( + checkLimits(usage(500, 500), { maxInputTokens: 100, maxTotalTokens: 100 })?.kind, + 'input', + ) +}) + +test('checkLimits treats 0 and negative caps as "no cap"', () => { + assert.equal(checkLimits(usage(10, 10), { maxTotalTokens: 0, maxInputTokens: -1 }), undefined) +}) + +test('resolveLimits folds in the legacy maxTotalTokens; limits.maxTotalTokens wins', () => { + assert.deepEqual(resolveLimits({ maxTotalTokens: 500 }), { maxTotalTokens: 500 }) + assert.deepEqual(resolveLimits({ maxTotalTokens: 500, limits: { maxTotalTokens: 100 } }), { + maxTotalTokens: 100, + }) + assert.deepEqual(resolveLimits({ limits: { maxInputTokens: 7 } }), { maxInputTokens: 7 }) +}) + +test('normalizeUsage fills every detail from the AI SDK shape', () => { + assert.deepEqual( + normalizeUsage({ + inputTokens: 100, + outputTokens: 40, + totalTokens: 140, + inputTokenDetails: { cacheReadTokens: 60, cacheWriteTokens: 5 }, + outputTokenDetails: { reasoningTokens: 25 }, + }), + { + inputTokens: 100, + outputTokens: 40, + totalTokens: 140, + reasoningTokens: 25, + cachedInputTokens: 60, + cacheWriteTokens: 5, + }, + ) + // Missing everything -> zeros, never NaN; total derived when absent. + assert.deepEqual(normalizeUsage(undefined).totalTokens, 0) + assert.equal(normalizeUsage({ inputTokens: 3, outputTokens: 4 }).totalTokens, 7) +}) + +test('addUsage sums every field, treating missing details as 0', () => { + assert.deepEqual(addUsage(usage(1, 2), { inputTokens: 3, outputTokens: 4, totalTokens: 7 }), { + inputTokens: 4, + outputTokens: 6, + totalTokens: 10, + reasoningTokens: 0, + cachedInputTokens: 0, + cacheWriteTokens: 0, + }) +}) + +test('limitsStopCondition sums the run so far + the steps of the current call', async () => { + const stops: string[] = [] + const stop = limitsStopCondition( + () => usage(10, 10), + { maxTotalTokens: 50 }, + (b) => stops.push(b.kind), + ) + const step = { usage: { inputTokens: 10, outputTokens: 5, totalTokens: 15 } } + assert.equal(await stop({ steps: [step] as never }), false) // 20 + 15 + assert.equal(await stop({ steps: [step, step] as never }), true) // 20 + 30 + assert.deepEqual(stops, ['total']) + assert.equal(sumStepUsage([step, step]).totalTokens, 30) +}) diff --git a/tests/mcp-connect.test.ts b/tests/mcp-connect.test.ts index a32cd34..567ddca 100644 --- a/tests/mcp-connect.test.ts +++ b/tests/mcp-connect.test.ts @@ -9,6 +9,7 @@ import { finishMcpOAuth, MemoryOAuthStore, } from '../src/mcp-oauth.ts' +import { isReadOnlyTool } from '../src/approval.ts' import { connectMcpServers, filterTools } from '../src/mcp.ts' import type { IMcpServerConfig } from '../src/types.ts' @@ -31,8 +32,15 @@ interface RpcMessage { interface MockOptions { requireAuth?: boolean - tools?: { name: string; description?: string; inputSchema?: unknown }[] + tools?: { + name: string + description?: string + inputSchema?: unknown + annotations?: Record + }[] failListTools?: boolean + /** Paginate tools/list: this many tools per page, cursor = next offset. */ + pageSize?: number /** Awaited before each MCP POST is answered — used to prove overlap. */ gate?: () => Promise onMcpRequest?: () => void @@ -73,6 +81,8 @@ const createMockServer = (opts: MockOptions = {}) => { failRegistration: false, /** Set to answer refresh_token grants with a transient 503. */ failRefresh: false, + /** tools/list requests answered, with the cursor each one carried. */ + listCursors: [] as (string | undefined)[], } const handleRpc = (msg: RpcMessage): unknown => { @@ -94,6 +104,20 @@ const createMockServer = (opts: MockOptions = {}) => { if (opts.failListTools) { return { jsonrpc: '2.0', id: msg.id, error: { code: -32000, message: 'list exploded' } } } + const cursor = (msg.params as { cursor?: string } | undefined)?.cursor + state.listCursors.push(cursor) + if (opts.pageSize) { + const start = Number(cursor ?? 0) + const end = start + opts.pageSize + return { + jsonrpc: '2.0', + id: msg.id, + result: { + tools: tools.slice(start, end), + ...(end < tools.length ? { nextCursor: String(end) } : {}), + }, + } + } return { jsonrpc: '2.0', id: msg.id, result: { tools } } } if (msg.method === 'tools/call') { @@ -848,3 +872,89 @@ test('FileOAuthStore: concurrent writes do not clobber each other', async () => assert.equal(await store.get(`k${i}`), `v${i}`) } }) + +// ── pagination / connect timeout / annotations ───────────────────────────── + +test('connectMcpServers: follows tools/list pagination, on connect and on refresh', async () => { + const tools = Array.from({ length: 7 }, (_, i) => ({ + name: `t${i}`, + inputSchema: { type: 'object' }, + })) + const mock = createMockServer({ tools, pageSize: 3 }) + const mcp = await connect({ docs: { url: MCP_URL, fetch: mock.fetchFn } }) + assert.deepEqual( + Object.keys(mcp.tools), + tools.map((t) => `docs__${t.name}`), + ) + assert.deepEqual(mock.state.listCursors, [undefined, '3', '6']) + + tools.push({ name: 't7', inputSchema: { type: 'object' } }) + await mcp.refreshServer('docs') + assert.equal(Object.keys(mcp.tools).length, 8) + assert.deepEqual(mock.state.listCursors.slice(3), [undefined, '3', '6']) + await mcp.close() +}) + +test('connectMcpServers: a server that never answers times out; the others still mount', async () => { + const hanging = createMockServer({ gate: () => new Promise(() => {}) }) + const healthy = createMockServer() + const logger = collect() + const started = Date.now() + const mcp = await connectMcpServers( + { + slow: { url: MCP_URL, fetch: hanging.fetchFn }, + docs: { url: MCP_URL, fetch: healthy.fetchFn }, + }, + logger.log as never, + 'agent-test', + undefined, + undefined, + undefined, + { connectTimeoutMs: 150 }, + ) + assert.ok(Date.now() - started < 2_000, 'the timeout bounded the connect') + assert.deepEqual(mcp.results[0], { + name: 'slow', + connected: false, + error: 'connect timed out after 150ms', + }) + assert.equal(mcp.results[1].connected, true) + assert.deepEqual(Object.keys(mcp.tools), ['docs__echo']) + assert.ok(logger.lines.some((m) => m.includes('slow: failed to connect - connect timed out'))) + await mcp.close() +}) + +test('connectMcpServers: a per-server connectTimeoutMs wins over the default', async () => { + const hanging = createMockServer({ gate: () => new Promise(() => {}) }) + const mcp = await connectMcpServers( + { slow: { url: MCP_URL, fetch: hanging.fetchFn, connectTimeoutMs: 50 } }, + silent as never, + 'agent-test', + undefined, + undefined, + undefined, + { connectTimeoutMs: 60_000 }, + ) + assert.equal(mcp.results[0].error, 'connect timed out after 50ms') + await mcp.close() +}) + +test('connectMcpServers: annotations.readOnlyHint marks the tool and its catalogue entry', async () => { + const mock = createMockServer({ + tools: [ + { name: 'get', inputSchema: { type: 'object' }, annotations: { readOnlyHint: true } }, + { name: 'put', inputSchema: { type: 'object' }, annotations: { destructiveHint: true } }, + ], + }) + const mcp = await connect({ docs: { url: MCP_URL, fetch: mock.fetchFn } }) + assert.deepEqual( + mcp.catalog.map((c) => [c.name, c.readOnly]), + [ + ['docs__get', true], + ['docs__put', false], + ], + ) + assert.equal(isReadOnlyTool(mcp.tools.docs__get), true) + assert.equal(isReadOnlyTool(mcp.tools.docs__put), false) + await mcp.close() +}) diff --git a/tests/skills.test.ts b/tests/skills.test.ts new file mode 100644 index 0000000..5d255c6 --- /dev/null +++ b/tests/skills.test.ts @@ -0,0 +1,184 @@ +import assert from 'node:assert/strict' +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import os from 'node:os' +import path from 'node:path' +import test from 'node:test' +import { + createSkillTools, + defineSkill, + loadSkillsFromDir, + parseSkillMarkdown, + renderActiveSkills, + renderSkillsIndex, +} from '../src/skills.ts' +import { runContext } from '../src/context.ts' +import type { IRunState } from '../src/internal.ts' +import type { AgentEvent } from '../src/types.ts' + +test('parseSkillMarkdown: plain frontmatter + body', () => { + const s = parseSkillMarkdown( + '---\nname: pdf-tools\ndescription: Work with PDF files\nlicense: MIT\n---\n# PDF\n\nUse pdftotext.\n', + ) + assert.deepEqual(s, { + name: 'pdf-tools', + description: 'Work with PDF files', + content: '# PDF\n\nUse pdftotext.', + }) +}) + +test('parseSkillMarkdown: quoted values, CRLF, BOM and a trailing comment', () => { + const s = parseSkillMarkdown( + "\uFEFF---\r\nname: \"quoted-name\"\r\ndescription: 'It''s single-quoted'\r\nversion: 1 # ignored\r\n---\r\nBody", + ) + assert.equal(s.name, 'quoted-name') + assert.equal(s.description, "It's single-quoted") + assert.equal(s.content, 'Body') + const dq = parseSkillMarkdown('---\nname: dq\ndescription: "Say \\"hi\\" twice"\n---\nx') + assert.equal(dq.description, 'Say "hi" twice') +}) + +test('parseSkillMarkdown: folded (>) and literal (|) block scalars', () => { + const folded = parseSkillMarkdown( + '---\nname: folded\ndescription: >\n Use when the user\n asks about invoices.\nmetadata:\n owner: me\n---\nBody', + ) + assert.equal(folded.description, 'Use when the user asks about invoices.') + const literal = parseSkillMarkdown( + '---\nname: literal\ndescription: |-\n line one\n line two\n---\nBody', + ) + assert.equal(literal.description, 'line one\nline two') +}) + +test('parseSkillMarkdown: clear errors for a missing frontmatter, name, description or bad name', () => { + assert.throws(() => parseSkillMarkdown('# no frontmatter'), /frontmatter/) + assert.throws(() => parseSkillMarkdown('---\ndescription: d\n---\nx'), /missing "name"/) + assert.throws(() => parseSkillMarkdown('---\nname: ok\n---\nx'), /missing "description"/) + assert.throws( + () => parseSkillMarkdown('---\nname: Bad_Name\ndescription: d\n---\nx'), + /skill name must match/, + ) +}) + +test('defineSkill validates and normalises bundled file paths', () => { + const s = defineSkill({ + name: 'x', + description: ' d ', + content: 'c', + files: [{ path: './ref/a.md', content: 'A' }], + }) + assert.equal(s.description, 'd') + assert.deepEqual(s.files, [{ path: 'ref/a.md', content: 'A' }]) + assert.throws( + () => + defineSkill({ + name: 'x', + description: 'd', + content: 'c', + files: [{ path: '../up', content: '' }], + }), + /inside the skill/, + ) + assert.throws(() => defineSkill({ name: 'x', description: '', content: 'c' }), /description/) +}) + +test('loadSkillsFromDir: SKILL.md per folder, text files bundled, big/binary skipped', async () => { + const dir = await mkdtemp(path.join(os.tmpdir(), 'skills-')) + try { + await mkdir(path.join(dir, 'b-skill', 'ref'), { recursive: true }) + await mkdir(path.join(dir, 'a-skill')) + await mkdir(path.join(dir, 'no-skill')) + await writeFile( + path.join(dir, 'b-skill', 'SKILL.md'), + '---\nname: b-skill\ndescription: B things\n---\nDo B.', + ) + await writeFile(path.join(dir, 'b-skill', 'ref', 'notes.md'), 'notes') + await writeFile(path.join(dir, 'b-skill', 'script.py'), 'print(1)') + await writeFile(path.join(dir, 'b-skill', 'image.png'), Buffer.from([1, 2, 3])) + await writeFile(path.join(dir, 'b-skill', 'huge.txt'), 'x'.repeat(257 * 1024)) + await writeFile( + path.join(dir, 'a-skill', 'SKILL.md'), + '---\nname: a-skill\ndescription: A things\n---\nDo A.', + ) + await writeFile(path.join(dir, 'no-skill', 'README.md'), 'ignored') + + const skills = await loadSkillsFromDir(dir) + assert.deepEqual( + skills.map((s) => s.name), + ['a-skill', 'b-skill'], + ) + assert.deepEqual(skills[1].files, [ + { path: 'ref/notes.md', content: 'notes' }, + { path: 'script.py', content: 'print(1)' }, + ]) + assert.equal(skills[0].files, undefined) + } finally { + await rm(dir, { recursive: true, force: true }) + } +}) + +test('renderSkillsIndex / renderActiveSkills (budget + clip marker)', () => { + const skills = [ + defineSkill({ name: 'a', description: 'Alpha\nskill', content: 'A'.repeat(50) }), + defineSkill({ name: 'b', description: 'Beta', content: 'B'.repeat(50) }), + ] + assert.equal(renderSkillsIndex(skills), '- a: Alpha skill\n- b: Beta') + const full = renderActiveSkills(skills, ['b', 'a']) + assert.ok(full.indexOf('### Skill: b') < full.indexOf('### Skill: a')) + const clipped = renderActiveSkills(skills, ['a', 'b'], 80) + assert.match(clipped, /\[skill instructions clipped/) +}) + +test('load_skill activates the skill in the run and lists its files; read_skill_file reads them', async () => { + const skills = [ + defineSkill({ + name: 'invoices', + description: 'Invoice rules', + content: 'Always add VAT.', + files: [{ path: 'rates.json', content: '{"vat":0.2}' }], + }), + ] + const tools = createSkillTools(skills) as Record< + string, + { execute: (input: unknown, opts: unknown) => Promise; readOnly?: boolean } + > + assert.equal(tools.load_skill.readOnly, true) + const events: AgentEvent[] = [] + const state: IRunState = { + usage: { inputTokens: 0, outputTokens: 0, totalTokens: 0 }, + toolCalls: 0, + strategy: 'all', + discovered: [], + activeSkills: [], + traceSummaryUpTo: 0, + } + await runContext.run( + { runId: 'r', startedAt: 0, sandboxDir: '/tmp/x', state, emit: (e) => events.push(e) }, + async () => { + const loaded = await tools.load_skill.execute({ name: 'invoices' }, {}) + assert.deepEqual(loaded, { + name: 'invoices', + content: 'Always add VAT.', + files: ['rates.json'], + }) + // Idempotent: a second load does not re-emit. + await tools.load_skill.execute({ name: 'invoices' }, {}) + assert.equal( + await tools.read_skill_file.execute({ name: 'invoices', path: './rates.json' }, {}), + '{"vat":0.2}', + ) + await assert.rejects( + () => tools.load_skill.execute({ name: 'nope' }, {}), + /Available skills: invoices/, + ) + await assert.rejects( + () => tools.read_skill_file.execute({ name: 'invoices', path: 'missing.md' }, {}), + /Files: rates.json/, + ) + }, + ) + assert.deepEqual(state.activeSkills, ['invoices']) + assert.deepEqual( + events.map((e) => e.type === 'skill.activated' && `${e.name}:${e.by}`), + ['invoices:tool'], + ) + assert.deepEqual(createSkillTools([]), {}) +}) diff --git a/tests/subagent.test.ts b/tests/subagent.test.ts new file mode 100644 index 0000000..defa421 --- /dev/null +++ b/tests/subagent.test.ts @@ -0,0 +1,72 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + assertWorkerSafeConfig, + slimEvent, + subagentWorkerUrl, + toCloneable, + workerExecArgv, +} from '../src/subagent.ts' +import type { IAgentConfig } from '../src/types.ts' + +const config = (extra: Partial = {}): IAgentConfig => ({ + clientName: 'c', + providerType: 'openai', + apiKey: 'k', + model: 'm', + mcpServers: {}, + maxIterations: 1, + maxStepsPerTask: 1, + logLevel: 'none', + ...extra, +}) + +test('assertWorkerSafeConfig names the first function-valued field', () => { + assert.doesNotThrow(() => assertWorkerSafeConfig(config({ thinking: 'high' }))) + assert.throws( + () => + assertWorkerSafeConfig( + config({ mcpServers: { docs: { url: 'https://x', getHeaders: () => ({}) } } }), + ), + /config\.mcpServers\.docs\.getHeaders is a function/, + ) + assert.throws( + () => + assertWorkerSafeConfig(config({ toolApproval: { mode: 'ask-all', onRequest: () => true } })), + /config\.toolApproval\.onRequest/, + ) +}) + +test('workerExecArgv keeps runtime flags and drops the entry-point ones', () => { + assert.deepEqual( + workerExecArgv([ + '--experimental-strip-types', + '--input-type=module', + '-e', + 'code()', + '--disable-warning=ExperimentalWarning', + '--input-type', + 'commonjs', + ]), + ['--experimental-strip-types', '--disable-warning=ExperimentalWarning'], + ) +}) + +test('toCloneable / slimEvent: Errors become messages only when cloning fails; long strings clipped', () => { + const withFn = { a: 1, f: () => 1, e: new Error('boom') } + assert.deepEqual(toCloneable(withFn), { a: 1, e: 'boom' }) + assert.deepEqual(toCloneable({ a: [1, 'x'] }), { a: [1, 'x'] }) + const slim = slimEvent({ + type: 'step.tool-result', + step: { id: 's', description: 'd', expectedOutcome: 'e' }, + name: 't', + output: 'z'.repeat(10_000), + ok: true, + }) as { output: string } + assert.ok(slim.output.length < 5_000) + assert.match(slim.output, /… \[truncated 6000 chars\]$/) +}) + +test('subagentWorkerUrl points at the .ts entry from source', () => { + assert.match(subagentWorkerUrl().pathname, /\/src\/subagent-worker\.ts$/) +}) diff --git a/tests/thinking.test.ts b/tests/thinking.test.ts new file mode 100644 index 0000000..53b42d9 --- /dev/null +++ b/tests/thinking.test.ts @@ -0,0 +1,144 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { buildInstructions, cacheProviderOptions, resolvePromptCaching } from '../src/caching.ts' +import { stageCallOptions } from '../src/call-options.ts' +import { + mergeProviderOptions, + resolveThinking, + stageEmitsThoughts, + stageThinkingSetting, +} from '../src/thinking.ts' +import type { IAgentConfig } from '../src/types.ts' + +test('resolveThinking: false / undefined send nothing', () => { + assert.deepEqual(resolveThinking(undefined), {}) + assert.deepEqual(resolveThinking(false), {}) +}) + +test("resolveThinking: true is 'medium' with thought streaming requested", () => { + assert.deepEqual(resolveThinking(true), { + reasoning: 'medium', + providerOptions: { + google: { thinkingConfig: { includeThoughts: true } }, + openai: { reasoningSummary: 'auto' }, + }, + }) +}) + +test('resolveThinking: a bare level becomes { level }', () => { + const r = resolveThinking('high') + assert.equal(r.reasoning, 'high') + assert.deepEqual(r.providerOptions?.openai, { reasoningSummary: 'auto' }) +}) + +test("resolveThinking: 'none' disables explicitly and sends no provider options", () => { + assert.deepEqual(resolveThinking('none'), { reasoning: 'none' }) + assert.deepEqual(resolveThinking({ level: 'none', budgetTokens: 2000 }), { reasoning: 'none' }) +}) + +test('resolveThinking: budgetTokens maps to Anthropic and Google budgets', () => { + assert.deepEqual(resolveThinking({ budgetTokens: 4096 }), { + reasoning: 'medium', + providerOptions: { + anthropic: { thinking: { type: 'enabled', budgetTokens: 4096 } }, + google: { thinkingConfig: { thinkingBudget: 4096, includeThoughts: true } }, + }, + }) + const quiet = resolveThinking({ level: 'low', budgetTokens: 1024, includeThoughts: false }) + assert.equal(quiet.reasoning, 'low') + assert.deepEqual(quiet.providerOptions?.google, { + thinkingConfig: { thinkingBudget: 1024, includeThoughts: false }, + }) +}) + +test('resolveThinking: includeThoughts false without a budget sends only the level', () => { + assert.deepEqual(resolveThinking({ level: 'xhigh', includeThoughts: false }), { + reasoning: 'xhigh', + }) +}) + +test('stageThinkingSetting: a stage entry (even false) wins over the top level', () => { + const config = { thinking: 'high' as const, stageThinking: { executor: false as const } } + assert.equal(stageThinkingSetting(config, 'planner'), 'high') + assert.equal(stageThinkingSetting(config, 'executor'), false) + assert.equal(stageEmitsThoughts({ thinking: { includeThoughts: false } }, 'synthesizer'), false) + assert.equal(stageEmitsThoughts({}, 'synthesizer'), true) +}) + +test('mergeProviderOptions deep-merges per provider key, later layers win', () => { + const merged = mergeProviderOptions( + { google: { thinkingConfig: { includeThoughts: true } }, openai: { reasoningSummary: 'auto' } }, + undefined, + { openai: { promptCacheKey: 'k' }, google: { thinkingConfig: { thinkingBudget: 5 } } }, + ) + assert.deepEqual(merged, { + google: { thinkingConfig: { includeThoughts: true, thinkingBudget: 5 } }, + openai: { reasoningSummary: 'auto', promptCacheKey: 'k' }, + }) + assert.equal(mergeProviderOptions(undefined, undefined), undefined) +}) + +test('resolvePromptCaching: default on, false off, options carried', () => { + assert.deepEqual(resolvePromptCaching(undefined), { enabled: true }) + assert.deepEqual(resolvePromptCaching(false), { enabled: false }) + assert.deepEqual(resolvePromptCaching({ ttl: '1h', key: 'x' }), { + enabled: true, + ttl: '1h', + key: 'x', + }) +}) + +test('buildInstructions: a cache breakpoint on the system message when caching is on', () => { + assert.equal(buildInstructions('SYS', { enabled: false }), 'SYS') + assert.deepEqual(buildInstructions('SYS', { enabled: true, ttl: '1h' }), { + role: 'system', + content: 'SYS', + providerOptions: { anthropic: { cacheControl: { type: 'ephemeral', ttl: '1h' } } }, + }) + assert.deepEqual(cacheProviderOptions({ enabled: true }, 'app', 'executor'), { + openai: { promptCacheKey: 'app:executor' }, + }) + assert.equal(cacheProviderOptions({ enabled: false }, 'app', 'executor'), undefined) +}) + +const config = (extra: Partial = {}): IAgentConfig => ({ + clientName: 'app', + providerType: 'openai', + apiKey: 'k', + model: 'm', + mcpServers: {}, + maxIterations: 1, + maxStepsPerTask: 1, + logLevel: 'none', + ...extra, +}) + +test('stageCallOptions merges thinking + caching and applies the per-call cap', () => { + const call = stageCallOptions( + config({ + thinking: 'low', + stageThinking: { planner: { budgetTokens: 2048 } }, + limits: { perCall: { planner: 900 } }, + promptCaching: { key: 'fixed' }, + }), + 'planner', + 'PLANNER SYSTEM', + ) + assert.equal(call.reasoning, 'medium') + assert.equal(call.maxOutputTokens, 900) + assert.deepEqual(call.providerOptions?.openai, { promptCacheKey: 'fixed' }) + assert.deepEqual(call.providerOptions?.anthropic, { + thinking: { type: 'enabled', budgetTokens: 2048 }, + }) + assert.equal(typeof call.instructions, 'object') + + // Defaults: no thinking, no cap, caching on. + const plain = stageCallOptions(config(), 'executor', 'EXEC') + assert.equal(plain.reasoning, undefined) + assert.equal(plain.maxOutputTokens, undefined) + assert.deepEqual(plain.providerOptions, { openai: { promptCacheKey: 'app:executor' } }) + + const uncached = stageCallOptions(config({ promptCaching: false }), 'synthesizer', 'S') + assert.equal(uncached.instructions, 'S') + assert.equal(uncached.providerOptions, undefined) +}) diff --git a/tests/token-efficiency.test.ts b/tests/token-efficiency.test.ts new file mode 100644 index 0000000..f808c56 --- /dev/null +++ b/tests/token-efficiency.test.ts @@ -0,0 +1,132 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { tool, type ModelMessage } from 'ai' +import { z } from 'zod' +import { + createAgent, + createToolResultClearer, + withRollingBreakpoint, + type AgentEvent, + type IAgentConfig, +} from '../src/index.ts' +import { + startLocalOpenAI, + systemOf, + toolNames, + toolResults, + type IChatRequest, + type Script, +} from './helpers/openai-server.ts' + +// The token-saving mechanics shared with the browser sibling: a rolling cache +// breakpoint in the tool loop, clearing stale tool results, a sorted tool list. + +const BREAKPOINT = { anthropic: { cacheControl: { type: 'ephemeral' } } } + +test('withRollingBreakpoint: only the newest message carries it; earlier ones lose theirs', () => { + const msgs = [ + { role: 'user', content: 'a' }, + { role: 'assistant', content: 'b', providerOptions: { openai: { x: 1 } } }, + ] + const out = withRollingBreakpoint(msgs, { enabled: true, ttl: '1h' }) + assert.equal(out[0].providerOptions, undefined) + assert.deepEqual(out[1].providerOptions, { + openai: { x: 1 }, + anthropic: { cacheControl: { type: 'ephemeral', ttl: '1h' } }, + }) + const next = withRollingBreakpoint([...out, { role: 'user', content: 'c' }], { enabled: true }) + assert.deepEqual(next[1].providerOptions, { openai: { x: 1 } }) + assert.deepEqual(next[2].providerOptions, BREAKPOINT) + assert.equal(withRollingBreakpoint(msgs, { enabled: false }), msgs) +}) + +test('createToolResultClearer: oldest results past the trigger become stubs, sticky, newest kept', () => { + const infos: unknown[] = [] + const clear = createToolResultClearer({ triggerTokens: 30, keep: 1 }, (i) => infos.push(i)) + const result = (id: string, value: string): ModelMessage => ({ + role: 'tool', + content: [ + { type: 'tool-result', toolCallId: id, toolName: 'read', output: { type: 'text', value } }, + ], + }) + const valueOf = (m: ModelMessage) => + (m.content as { output: { value: string } }[])[0].output.value + const first: ModelMessage[] = [{ role: 'user', content: 'go' }, result('a', 'small')] + assert.equal(clear(first), first) + const big = 'x'.repeat(200) + const edited = clear([...first, result('b', big), result('c', big)]) + assert.match(valueOf(edited[1]), /read result cleared/) + assert.match(valueOf(edited[2]), /read result cleared/) + assert.equal(valueOf(edited[3]), big) + assert.equal(infos.length, 1) + assert.match(valueOf(clear(first)[1]), /cleared/) // sticky +}) + +const isPlanner = (req: IChatRequest) => systemOf(req).includes('You are the Planner') +const isSynth = (req: IChatRequest) => systemOf(req).includes('You are the Synthesizer') + +const config = (baseURL: string, extra: Partial): IAgentConfig => ({ + clientName: 'token-test', + providerType: 'openai-compatible', + baseURL, + apiKey: 'not-a-real-key', + model: 'scripted', + mcpServers: {}, + maxIterations: 3, + maxStepsPerTask: 6, + logLevel: 'none', + ...extra, +}) + +test('tool loop: stale results are cleared on the wire, reported, and tools go in name order', async () => { + const big = 'y'.repeat(4000) + const executorRequests: IChatRequest[] = [] + const script: Script = (req) => { + if (isPlanner(req)) { + return { + text: JSON.stringify({ + thought: 'read', + steps: [{ id: 's1', description: 'Read three pages', expectedOutcome: 'read' }], + }), + } + } + if (isSynth(req)) return { text: 'Read them all.' } + executorRequests.push(req) + const n = toolResults(req).length + return n < 3 ? { toolCalls: [{ name: 'page', args: { n } }] } : { text: 'Read three pages.' } + } + const llm = await startLocalOpenAI(script) + const events: AgentEvent[] = [] + const agent = await createAgent( + config(llm.baseURL, { + compaction: { clearToolResultsAfterTokens: 1200, keepToolResults: 1 }, + tools: { + zeta: tool({ description: 'z', inputSchema: z.object({}), execute: async () => 'z' }), + page: tool({ + description: 'Read a page', + inputSchema: z.object({ n: z.number() }), + execute: async () => big, + }), + }, + }), + ) + try { + await agent.run({ input: 'read', onEvent: (e: AgentEvent) => events.push(e) }) + const last = executorRequests.at(-1)! + const results = toolResults(last) + assert.equal(results.length, 3) + assert.match(results[0], /page result cleared to save context/) + assert.match(results[1], /page result cleared to save context/) + + assert.ok(results[2].includes(big)) + assert.ok( + events.some((e) => e.type === 'context.compacted' && e.scope === 'tool-results'), + 'the clearing is reported', + ) + const names = toolNames(executorRequests[0]) + assert.deepEqual(names, [...names].sort()) + } finally { + await agent.close() + await llm.close() + } +}) diff --git a/tests/tool-search.test.ts b/tests/tool-search.test.ts new file mode 100644 index 0000000..4212a8e --- /dev/null +++ b/tests/tool-search.test.ts @@ -0,0 +1,99 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + renderSearchCatalog, + resolveToolStrategy, + searchTools, + tokenizeForSearch, +} from '../src/tool-search.ts' + +const catalog = [ + { name: 'github__list_issues', description: 'List issues of a repository', server: 'github' }, + { name: 'github__create_issue', description: 'Open a new issue', server: 'github' }, + { name: 'jira__searchIssues', description: 'JQL search over tickets', server: 'jira' }, + { name: 'weather__forecast', description: 'Weather forecast for a city', server: 'weather' }, + { + name: 'files__read', + description: 'Read a file from the issue tracker export', + server: 'files', + }, +] + +test('tokenizeForSearch splits on non-alphanumerics and camelCase, lowercased', () => { + assert.deepEqual(tokenizeForSearch('jira__searchIssues'), ['jira', 'search', 'issues']) + assert.deepEqual(tokenizeForSearch('getHTTPResponse v2'), ['get', 'http', 'response', 'v2']) +}) + +test('searchTools scores 3·name + 2·server + 1·description per query term', () => { + const names = searchTools(catalog, 'issue').map((t) => t.name) + // name + description (4) beat name only (3) beat a description-only hit + // (1); ties keep catalogue order. + assert.deepEqual(names, [ + 'github__list_issues', + 'github__create_issue', + 'jira__searchIssues', + 'files__read', + ]) + // A longer term does not prefix-match a shorter token. + assert.deepEqual( + searchTools(catalog, 'issues').map((t) => t.name), + ['github__list_issues', 'jira__searchIssues'], + ) +}) + +test('searchTools: prefix matching for 3+ char terms, exact match for shorter ones', () => { + assert.deepEqual( + searchTools(catalog, 'forec').map((t) => t.name), + ['weather__forecast'], + ) + // "is" is too short to prefix-match "issue(s)". + assert.deepEqual(searchTools(catalog, 'is'), []) +}) + +test('searchTools: server filter, zero-score exclusion and limit clamping', () => { + assert.deepEqual( + searchTools(catalog, 'issue', { server: 'jira' }).map((t) => t.name), + ['jira__searchIssues'], + ) + assert.deepEqual(searchTools(catalog, 'nothing-matches-this'), []) + assert.equal(searchTools(catalog, 'issue', { limit: 1 }).length, 1) + const many = Array.from({ length: 40 }, (_, i) => ({ + name: `t${i}__thing`, + description: 'thing', + server: 's', + })) + assert.equal(searchTools(many, 'thing', { limit: 500 }).length, 20) + assert.equal(searchTools(many, 'thing').length, 8) +}) + +test('searchTools: a server-name term scores on server too', () => { + const ranked = searchTools(catalog, 'github issue') + assert.equal(ranked[0].name, 'github__list_issues') + assert.equal(ranked[1].name, 'github__create_issue') +}) + +test("resolveToolStrategy: 'auto' is 'all' up to the threshold, 'search' above", () => { + assert.equal(resolveToolStrategy({}, 40), 'all') + assert.equal(resolveToolStrategy({}, 41), 'search') + assert.equal(resolveToolStrategy({ toolSearchThreshold: 5 }, 6), 'search') + assert.equal(resolveToolStrategy({ toolSelectionStrategy: 'all' }, 500), 'all') + assert.equal(resolveToolStrategy({ toolSelectionStrategy: 'plan-narrowed' }, 1), 'plan-narrowed') +}) + +test('renderSearchCatalog groups by server within a budget and points at find_tools', () => { + const out = renderSearchCatalog(catalog) + assert.match(out, /^\[github\]\n- github__list_issues: List issues of a repository/) + assert.ok(out.includes('[weather]')) + assert.ok(!out.includes('more tools')) + + const big = Array.from({ length: 500 }, (_, i) => ({ + name: `srv__tool_${i}`, + description: 'x'.repeat(200), + server: 'srv', + })) + const clipped = renderSearchCatalog(big, 2_000) + assert.ok(clipped.length < 2_200) + assert.match(clipped, /… \d+ more tools — the executor can find them with find_tools$/) + // Descriptions are cut to 60 chars. + assert.ok(clipped.split('\n')[1].endsWith('x'.repeat(60))) +}) diff --git a/tsup.config.ts b/tsup.config.ts index 59b441f..9ee8bee 100644 --- a/tsup.config.ts +++ b/tsup.config.ts @@ -25,6 +25,19 @@ export default defineConfig([ entry: { index: 'src/index.ts' }, format: ['esm', 'cjs'], dts: true, + // The subagent tool resolves its worker entry with + // `new URL('./subagent-worker.js', import.meta.url)`; the shim gives the + // CJS build a real import.meta.url (pathToFileURL(__filename)). + shims: true, + }, + { + ...shared, + // Worker-thread entry of createSubagentTool({ isolation: 'worker' }). + // ESM only (package "type": "module"); every build of the library - + // ESM, CJS and the CLI bundle - resolves it next to itself in dist/. + clean: false, + entry: { 'subagent-worker': 'src/subagent-worker.ts' }, + format: ['esm'], }, { ...shared, From 8806370ea6493ef8c3dbbff51e69f5c0ac7cc89a Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 00:16:04 +0000 Subject: [PATCH 2/2] fix: Node 24 worker execArgv, live-model regression; release 0.0.31 - Subagent workers inherit the parent's node options by default; an explicit execArgv is passed only to strip entry-point flags (-e / -p / --input-type). Passing process.execArgv explicitly failed on Node 24 under `node --test` ("Initiated Worker with invalid execArgv flags"). - The real-model gate (Qwen2.5-3B) regressed: the rewritten planner rule put the canned "Answer the user directly" next to the lookup rule, extra executor rules made the 3B echo a guessed answer alongside the real lookup, and sorting tools by name reordered what the server declared. Reproduced locally with the pinned llama.cpp image and weights, fresh server, two attempts (as CI): planner rules restored with the autonomy rule added as rule 6, executor prompt as measured, tools in declaration order (servers already mount in declaration order, so the cached prefix stays stable). Now 2/2 on a fresh server. - Version 0.0.31. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01A7jHa5Pim4ufrdgxg561G5 --- README.md | 4 ++-- package-lock.json | 4 ++-- package.json | 2 +- src/agent.ts | 16 +++++-------- src/executor.ts | 9 ++------ src/prompts.ts | 41 ++++++++++++++++++++-------------- src/subagent.ts | 27 +++++++++++++++++----- tests/subagent.test.ts | 9 ++++++++ tests/token-efficiency.test.ts | 8 ++++--- 9 files changed, 72 insertions(+), 48 deletions(-) diff --git a/README.md b/README.md index aa2577c..a5524de 100644 --- a/README.md +++ b/README.md @@ -354,7 +354,7 @@ A denied call throws `ToolDeniedError` inside the tool, so the step records `ok: promptCaching: { ttl: '1h', key: 'my-app' } // or false to send plain system strings ``` -Inside a step's tool loop a second Anthropic breakpoint **rolls to the newest message** every round, so each round reads all earlier rounds from the cache and writes only the new tail (`withRollingBreakpoint`; earlier message breakpoints are removed, so a request carries at most two — Anthropic allows four). Tools are sent and catalogued **sorted by name**, so the tool list that heads every cached prefix is identical however the MCP servers connected. +Inside a step's tool loop a second Anthropic breakpoint **rolls to the newest message** every round, so each round reads all earlier rounds from the cache and writes only the new tail (`withRollingBreakpoint`; earlier message breakpoints are removed, so a request carries at most two — Anthropic allows four). Tools keep **declaration order** — servers connect concurrently but mount in the order they are configured — so the tool list that heads every cached prefix is identical from run to run without reordering what a server declared (a small model was measured to depend on that order). Usage reports `cachedInputTokens` / `cacheWriteTokens`, so the effect is measurable. @@ -364,7 +364,7 @@ The practices coding agents (Claude Code, the Claude and GitHub Copilot extensio | Practice | How | Knob | | --- | --- | --- | -| Stable, cacheable prefix | run-stable system prompts, sorted tools, dynamic content last | `promptCaching` | +| Stable, cacheable prefix | run-stable system prompts, deterministic tool order, dynamic content last | `promptCaching` | | Cache the growing loop | rolling Anthropic breakpoint; OpenAI `promptCacheKey` per stage | `promptCaching: { ttl }` | | Don't send every tool | compact catalogue + `find_tools` above the threshold (deferred tools) | `toolSelectionStrategy`, `toolSearchThreshold` | | Load instructions on demand | skills: name + description in the prompt, body on activation | `skills` | diff --git a/package-lock.json b/package-lock.json index b0831e7..7bdece3 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@dudko.dev/agent", - "version": "0.0.30", + "version": "0.0.31", "lockfileVersion": 7, "requires": true, "packages": { "": { "name": "@dudko.dev/agent", - "version": "0.0.30", + "version": "0.0.31", "funding": [ { "type": "individual", diff --git a/package.json b/package.json index 4d79c66..09ef612 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@dudko.dev/agent", - "version": "0.0.30", + "version": "0.0.31", "type": "module", "description": "Tool-using planning agent over MCP servers, built on the Vercel AI SDK.", "keywords": [ diff --git a/src/agent.ts b/src/agent.ts index 0290ea2..7dc3510 100644 --- a/src/agent.ts +++ b/src/agent.ts @@ -334,17 +334,13 @@ export const createAgent = async ( throw err } - // Sorted by name, so the catalogue rendered into the (cached) planner prompt - // is the same whatever order the servers answered in. const toEntries = (catalog: typeof connected.catalog): IToolCatalogEntry[] => - catalog - .map((c) => ({ - name: c.name, - description: c.description, - server: c.server, - readOnly: c.readOnly === true, - })) - .sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)) + catalog.map((c) => ({ + name: c.name, + description: c.description, + server: c.server, + readOnly: c.readOnly === true, + })) const ctx: IAgentInternalContext = { config, diff --git a/src/executor.ts b/src/executor.ts index 43d602d..44b3dc7 100644 --- a/src/executor.ts +++ b/src/executor.ts @@ -114,11 +114,6 @@ export const executeStep = async ( }, ) -// The tool set with its keys in name order: the tool list heads every cached -// prompt prefix, so it must be identical however the MCP servers connected. -export const sortTools = >(tools: T): T => - Object.fromEntries(Object.entries(tools).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) as T - // Tools for one executor call: what the strategy exposes + the built-ins. // In 'search' mode the full catalogue is passed and `prepareStep` narrows // each LLM step to the active set (find_tools grows it mid-call). @@ -141,10 +136,10 @@ export const buildExecutorTools = ( active.add(name) } } - return { tools: sortTools(tools), active } + return { tools, active } } const base = buildActiveToolSet(ctx, step) - const tools: ToolSet = sortTools(Object.keys(builtins).length ? { ...base, ...builtins } : base) + const tools: ToolSet = Object.keys(builtins).length ? { ...base, ...builtins } : base // Defence-in-depth: explicitly tell the SDK which tools are callable in // this step. Only worth it in narrowed mode; in 'all' mode it's just a // copy of every key, equivalent to omitting the field. diff --git a/src/prompts.ts b/src/prompts.ts index 77318a3..12223e2 100644 --- a/src/prompts.ts +++ b/src/prompts.ts @@ -166,19 +166,22 @@ export const withDomainContext = (base: string, systemPrompt: string | undefined export const DEFAULT_PLAN_STEP_CAP = 8 -const plannerBase = ( - maxSteps: number, -): string => `You are the Planner of an autonomous multi-step agent system. +// The rule wording is load-bearing for small models: tests/live-model.test.ts +// runs a 3B against it. A canned "Answer the user directly" mentioned next to +// "never answer from memory" pulled the 3B into the answer-directly branch +// (it then invented the secret), so rule 1 keeps that branch for trivial +// tasks only and rule 4 sends questions the tools can answer to the tools. +const plannerBase = (maxSteps: number): string => `You are the Planner of a multi-step agent system. -Your only job is to decompose the user's request into a short ordered list of concrete, actionable steps that a tool-using Executor can carry out one at a time, on its own, without asking the user anything. +Your only job is to decompose the user's request into a short ordered list of concrete actionable steps that a tool-using Executor can perform one at a time. Rules: -1. Produce 1-5 steps for typical requests (hard cap is ${maxSteps}). Prefer FEWER, larger steps over many micro-steps. +1. Produce 1-5 steps for typical requests (hard cap is ${maxSteps}). Prefer FEWER, larger steps over many micro-steps. If the task is trivial and needs no tools, output a single step "Answer the user directly". 2. Each step must be self-contained, action-oriented, and verifiable. State what should be done and what the expected outcome is. -3. Plan tool use whenever the available tools can obtain, check or act on what the request needs. A question that needs data the tools can retrieve IS actionable: plan the lookups, never answer it from memory. Only when no tool is relevant (greetings, thanks, small talk, or a question fully answerable from the conversation or general knowledge) output a single step "Answer the user directly". -4. If a step needs a tool, suggest tool name(s) ONLY from the provided available-tools list. NEVER fabricate tool names. -5. Never plan a step that asks the user for clarification, confirmation or permission - the agent runs autonomously and the host handles tool consent. If the request is ambiguous, pick the most reasonable interpretation and state the assumption in the step description. -6. The last step must produce the deliverable for the user (do not append a separate "summarize" step - the system synthesizes the final answer). +3. If a step needs a tool, suggest tool name(s) ONLY from the provided available-tools list. NEVER fabricate tool names. +4. If the request asks for information that is likely retrievable via the available tools, plan to use them. If no tool fits, plan to answer from general knowledge. +5. The last step must produce the deliverable for the user (do not append a separate "summarize" step - the system synthesizes the final answer). +6. The agent works autonomously: never plan a step that asks the user for clarification, confirmation or permission (the host handles tool consent). If the request is ambiguous, pick the most reasonable interpretation and state the assumption in the step description. 7. Output strict JSON matching the requested schema. No prose outside JSON.` export const PLANNER_SYSTEM_BASE = plannerBase(DEFAULT_PLAN_STEP_CAP) @@ -255,16 +258,20 @@ export const buildPlannerUserPrompt = (input: string, history?: IConversationTur .filter(Boolean) .join('\n') -export const EXECUTOR_SYSTEM = `You are the Executor of an autonomous multi-step agent system. You receive ONE step at a time and you must accomplish only that step. +// The executor's wording is measured, not tuned by taste: on the 3B the +// release gate runs (tests/live-model.test.ts) any extra rule here - even a +// one-line "never ask the user" - made it call the lookup AND echo a guessed +// answer in parallel, then report the guess. Autonomy is carried by the +// planner, replanner and synthesizer rules and by [BLOCKER] → replanner; +// re-run the live test before touching this text. +export const EXECUTOR_SYSTEM = `You are the Executor of a multi-step agent system. You receive ONE step at a time and you must accomplish only that step. Rules: -1. Stay focused on the CURRENT step. Do not jump ahead, do not redo finished steps. Read prior step results in the trace before re-fetching the same data. -2. Act on your own. Look things up with the available tools instead of guessing. Never ask the user questions or for confirmation - nobody answers mid-run, and tool consent is handled by the host: just call the tool you need. -3. When details are missing, choose sensible defaults and state the assumptions you made in the step result. -4. Call tools with valid arguments. If a call fails, fix the arguments or try another tool before giving up. If a tool call is denied, do not retry it; continue without it. -5. When the step is complete, write a short concrete "step result" describing what you found / did. Include identifiers, names, or key data the next step might need. Do not fabricate data. -6. Only when the step is truly impossible (missing credentials or permissions, a denied tool, data that does not exist or cannot be reached), explain the blocker briefly and end your reply with the literal token [BLOCKER] on its own line. The system uses this token (language-independent) to invoke the Replanner. -7. Be concise. Do not narrate your reasoning at length - the Replanner reads only your final summary.` +1. Stay focused on the CURRENT step. Do not jump ahead, do not redo finished steps. +2. Use the available tools when they help. Call them with valid arguments. Read prior step results in the trace before re-fetching the same data. +3. When the step is complete, write a short concrete "step result" describing what you found / did. Include identifiers, names, or key data the next step might need. +4. If the step is impossible with the available tools, or if you are otherwise blocked, explain the blocker briefly and end your reply with the literal token [BLOCKER] on its own line. The system uses this token (language-independent) to invoke the Replanner. Do not fabricate data. +5. Be concise. Do not narrate your reasoning at length - the Replanner reads only your final summary.` const EXECUTOR_SEARCH_NOTE = ` diff --git a/src/subagent.ts b/src/subagent.ts index dbcdd75..13e00df 100644 --- a/src/subagent.ts +++ b/src/subagent.ts @@ -142,11 +142,25 @@ export const subagentWorkerUrl = (): URL => { return new URL(here.endsWith('.ts') ? './subagent-worker.ts' : './subagent-worker.js', here) } -// The parent's node flags minus the ones that describe ITS entry point -// (-e / -p / --input-type would make the worker misread its own file). -// Everything else - notably --experimental-strip-types for source runs - is -// inherited. -export const workerExecArgv = (argv: readonly string[] = process.execArgv): string[] => { +// The worker's node flags. By default (undefined) a worker inherits the +// parent's options itself - notably --experimental-strip-types for source runs +// - and Node drops the per-process ones a worker can't take; passing +// process.execArgv explicitly instead fails on Node 24 under `node --test` +// ("Initiated Worker with invalid execArgv flags: --stack-trace-limit…"). +// Only when the parent's flags describe ITS entry point (-e / -p / +// --input-type would make the worker misread its own file) is an explicit, +// filtered list needed. +export const workerExecArgv = ( + argv: readonly string[] = process.execArgv, +): string[] | undefined => { + const entryPoint = argv.some( + (a) => + ['-e', '--eval', '-p', '--print', '--input-type'].includes(a) || + /^--(input-type|eval|print)=/.test(a), + ) + if (!entryPoint) { + return undefined + } const out: string[] = [] for (let i = 0; i < argv.length; i++) { const a = argv[i] @@ -273,9 +287,10 @@ const runInWorker = ( let settled = false let worker: Worker try { + const execArgv = workerExecArgv() worker = new Worker(subagentWorkerUrl(), { workerData: { subagent: opts.name }, - execArgv: workerExecArgv(), + ...(execArgv ? { execArgv } : {}), }) } catch (err) { reject(err) diff --git a/tests/subagent.test.ts b/tests/subagent.test.ts index defa421..ba9177c 100644 --- a/tests/subagent.test.ts +++ b/tests/subagent.test.ts @@ -52,6 +52,15 @@ test('workerExecArgv keeps runtime flags and drops the entry-point ones', () => ) }) +test('workerExecArgv: without entry-point flags the worker inherits the options itself', () => { + // An explicit list breaks on Node 24 under `node --test`, whose execArgv + // carries per-process flags a worker rejects. + assert.equal( + workerExecArgv(['--experimental-strip-types', '--stack-trace-limit=10', '--v8-pool-size=4']), + undefined, + ) +}) + test('toCloneable / slimEvent: Errors become messages only when cloning fails; long strings clipped', () => { const withFn = { a: 1, f: () => 1, e: new Error('boom') } assert.deepEqual(toCloneable(withFn), { a: 1, e: 'boom' }) diff --git a/tests/token-efficiency.test.ts b/tests/token-efficiency.test.ts index f808c56..7339bc5 100644 --- a/tests/token-efficiency.test.ts +++ b/tests/token-efficiency.test.ts @@ -78,7 +78,7 @@ const config = (baseURL: string, extra: Partial): IAgentConfig => ...extra, }) -test('tool loop: stale results are cleared on the wire, reported, and tools go in name order', async () => { +test('tool loop: stale results are cleared on the wire, reported; tools keep declaration order', async () => { const big = 'y'.repeat(4000) const executorRequests: IChatRequest[] = [] const script: Script = (req) => { @@ -123,8 +123,10 @@ test('tool loop: stale results are cleared on the wire, reported, and tools go i events.some((e) => e.type === 'context.compacted' && e.scope === 'tool-results'), 'the clearing is reported', ) - const names = toolNames(executorRequests[0]) - assert.deepEqual(names, [...names].sort()) + // Declaration order, not sorted: servers mount in declaration order (so the + // cached prefix is stable already), and a server's own order is what a + // small model was measured against. + assert.deepEqual(toolNames(executorRequests[0]), ['zeta', 'page']) } finally { await agent.close() await llm.close()