chore: generate

This commit is contained in:
opencode-agent[bot]
2026-07-03 04:49:44 +00:00
parent cb93114424
commit 83c638eaac
20 changed files with 2137 additions and 1207 deletions
+9 -8
View File
@@ -55,7 +55,9 @@ const runtime = CodeMode.make({
}, },
}) })
const result = yield* runtime.execute(` const result =
yield *
runtime.execute(`
const order = await tools.orders.lookup({ id: "order_42" }) const order = await tools.orders.lookup({ id: "order_42" })
return { id: order.id, needsAttention: order.status !== "complete" } return { id: order.id, needsAttention: order.status !== "complete" }
`) `)
@@ -89,7 +91,9 @@ The description and schemas are part of the model-visible tool contract. Keep de
Use `CodeMode.execute` for a single execution: Use `CodeMode.execute` for a single execution:
```ts ```ts
const result = yield* CodeMode.execute({ const result =
yield *
CodeMode.execute({
tools: { orders: { lookup: lookupOrder } }, tools: { orders: { lookup: lookupOrder } },
code: `return await tools.orders.lookup({ id: "order_42" })`, code: `return await tools.orders.lookup({ id: "order_42" })`,
limits: { maxToolCalls: 10 }, limits: { maxToolCalls: 10 },
@@ -223,7 +227,7 @@ CodeMode is an orchestration language, not a general JavaScript runtime.
The limits are exactly three knobs: The limits are exactly three knobs:
| Limit | Default | Bounds | | Limit | Default | Bounds |
| --- | ---: | --- | | ---------------- | -------------------: | -------------------------------------------------------------------- |
| `timeoutMs` | none — no timeout | Wall-clock execution time. | | `timeoutMs` | none — no timeout | Wall-clock execution time. |
| `maxToolCalls` | none — unlimited | Tool calls admitted during the execution. | | `maxToolCalls` | none — unlimited | Tool calls admitted during the execution. |
| `maxOutputBytes` | none — no truncation | Model-facing output: the serialized result value plus captured logs. | | `maxOutputBytes` | none — no truncation | Model-facing output: the serialized result value plus captured logs. |
@@ -255,7 +259,7 @@ Two interpreter internals are fixed constants rather than knobs: at most 8 tool
Failures are data: Failures are data:
| Kind | Meaning | | Kind | Meaning |
| --- | --- | | ----------------------- | -------------------------------------------------------------------------------------------------------- |
| `ParseError` | Source is empty or cannot be parsed. | | `ParseError` | Source is empty or cannot be parsed. |
| `UnsupportedSyntax` | Parsed JavaScript is outside the supported subset. | | `UnsupportedSyntax` | Parsed JavaScript is outside the supported subset. |
| `UnknownTool` | A program referenced a tool the host did not provide. | | `UnknownTool` | A program referenced a tool the host did not provide. |
@@ -272,10 +276,7 @@ Unknown host failures, defects, invalid outputs, and copying failures are saniti
```ts ```ts
import { toolError } from "@opencode-ai/codemode" import { toolError } from "@opencode-ai/codemode"
run: ({ id }) => run: ({ id }) => (authorized(id) ? loadOrder(id) : Effect.fail(toolError("Order is unavailable")))
authorized(id)
? loadOrder(id)
: Effect.fail(toolError("Order is unavailable"))
``` ```
Only the supplied message is model-visible. The optional cause is never returned in `ExecuteResult`; hosts should perform any required internal logging before crossing this boundary. Only the supplied message is model-visible. The optional cause is never returned in `ExecuteResult`; hosts should perform any required internal logging before crossing this boundary.
+36 -6
View File
@@ -39,6 +39,7 @@ package and was **deleted** in Wave 3 (done, see below).
From issue #34787 and design discussion. Do not relitigate these casually. From issue #34787 and design discussion. Do not relitigate these casually.
### Core direction ### Core direction
- Generic CodeMode lives in its own package: `@opencode-ai/codemode` (repo scope convention; - Generic CodeMode lives in its own package: `@opencode-ai/codemode` (repo scope convention;
the issue's `@opencode/codemode` name was normalized to the `@opencode-ai/*` convention). the issue's `@opencode/codemode` name was normalized to the `@opencode-ai/*` convention).
- **Keep the hand-rolled interpreter.** No QuickJS/V8/sandbox-engine dependency. We own and - **Keep the hand-rolled interpreter.** No QuickJS/V8/sandbox-engine dependency. We own and
@@ -52,6 +53,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
products/blog posts) in code, comments, commit messages, or docs in this repo. products/blog posts) in code, comments, commit messages, or docs in this repo.
### MCP / tools ### MCP / tools
- The MCP adapter lives in OpenCode, not here. It converts MCP definitions into ordinary - The MCP adapter lives in OpenCode, not here. It converts MCP definitions into ordinary
`Tool.make(...)` definitions and hands CodeMode a plain tool tree. `Tool.make(...)` definitions and hands CodeMode a plain tool tree.
- Permissions stay in the OpenCode adapter (each tool's `run` wraps the permission ask). - Permissions stay in the OpenCode adapter (each tool's `run` wraps the permission ask).
@@ -61,6 +63,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
`tools.<server>.<tool>` namespaces before handing them over. `tools.<server>.<tool>` namespaces before handing them over.
### Discovery / search ### Discovery / search
- **Search only — no separate `describe`.** `tools.$codemode.search({ query?, namespace?, - **Search only — no separate `describe`.** `tools.$codemode.search({ query?, namespace?,
limit? })` over the final tool tree, owned by this package. limit? })` over the final tool tree, owned by this package.
- Search result item shape: `{ path, description, signature }` in an `{ items, total }` - Search result item shape: `{ path, description, signature }` in an `{ items, total }`
@@ -82,6 +85,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
- Tools without an output schema render `unknown` as their return type. - Tools without an output schema render `unknown` as their return type.
### Schemas / Tool.make ### Schemas / Tool.make
- `Tool.make` carries rich metadata so search can render real signatures. - `Tool.make` carries rich metadata so search can render real signatures.
- Support **Effect Schema** (first-class, validating) and **JSON Schema** (initially - Support **Effect Schema** (first-class, validating) and **JSON Schema** (initially
render-only — used for TypeScript rendering; the adapter may validate on its own). Leave render-only — used for TypeScript rendering; the adapter may validate on its own). Leave
@@ -90,6 +94,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
normalization for plugin authors can come later. normalization for plugin authors can come later.
### Attachments / output ### Attachments / output
- **No `output.text/file/image` API in v1.** (Deleted in Wave 2.) - **No `output.text/file/image` API in v1.** (Deleted in Wave 2.)
- Tool calls return native structured payloads into the sandbox. Files/images emitted by - Tool calls return native structured payloads into the sandbox. Files/images emitted by
child tools **never enter the sandbox** — the OpenCode adapter strips and accumulates them child tools **never enter the sandbox** — the OpenCode adapter strips and accumulates them
@@ -100,6 +105,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
image bytes into context or drop attachments. image bytes into context or drop attachments.
### Runtime behavior ### Runtime behavior
- Limits are EXACTLY the three public knobs: `{ timeoutMs, maxToolCalls, maxOutputBytes }` — - Limits are EXACTLY the three public knobs: `{ timeoutMs, maxToolCalls, maxOutputBytes }` —
matching the original locked spec exactly. NO limit has a default (user direction, Fix 6 matching the original locked spec exactly. NO limit has a default (user direction, Fix 6
for the first two; extended to `maxOutputBytes` in the truncation-layering fix below): for the first two; extended to `maxOutputBytes` in the truncation-layering fix below):
@@ -148,6 +154,7 @@ and `bun run typecheck`; from `packages/opencode`, `bun run typecheck` and
`test/tool/registry.test.ts`). `test/tool/registry.test.ts`).
### Wave 0 — scaffold (done) ### Wave 0 — scaffold (done)
- `packages/codemode` created from the experiments implementation: `src/{index,codemode,tool, - `packages/codemode` created from the experiments implementation: `src/{index,codemode,tool,
tool-error,tool-runtime}.ts`, README, AGENTS.md, tests. tool-error,tool-runtime}.ts`, README, AGENTS.md, tests.
- `package.json`: name `@opencode-ai/codemode`, deps `acorn@8.15.0`, `typescript: catalog:`, - `package.json`: name `@opencode-ai/codemode`, deps `acorn@8.15.0`, `typescript: catalog:`,
@@ -157,6 +164,7 @@ and `bun run typecheck`; from `packages/opencode`, `bun run typecheck` and
Context.Service key string renamed to `@opencode-ai/codemode/CurrentToolCall`. Context.Service key string renamed to `@opencode-ai/codemode/CurrentToolCall`.
### Wave 1a — forgiving JS semantics (done) ### Wave 1a — forgiving JS semantics (done)
Ported from the old opencode rune work; `test/parity.test.ts` (24 tests) is the acceptance Ported from the old opencode rune work; `test/parity.test.ts` (24 tests) is the acceptance
spec. The seeded interpreter was deliberately strict; these behaviors replaced that: spec. The seeded interpreter was deliberately strict; these behaviors replaced that:
@@ -173,6 +181,7 @@ spec. The seeded interpreter was deliberately strict; these behaviors replaced t
null/undefined still throws (real JS throws too). null/undefined still throws (real JS throws too).
### Wave 1b-i — stdlib value types: Date, RegExp, Map, Set (done) ### Wave 1b-i — stdlib value types: Date, RegExp, Map, Set (done)
`src/values.ts` holds `SandboxDate/SandboxRegExp/SandboxMap/SandboxSet` (own module so both `src/values.ts` holds `SandboxDate/SandboxRegExp/SandboxMap/SandboxSet` (own module so both
`codemode.ts` and `tool-runtime.ts` import without a cycle). Design: `codemode.ts` and `tool-runtime.ts` import without a cycle). Design:
@@ -202,13 +211,14 @@ spec. The seeded interpreter was deliberately strict; these behaviors replaced t
interpolation renders `/regex/` and ISO dates directly. interpolation renders `/regex/` and ISO dates directly.
### Wave 2 — API layer (done) ### Wave 2 — API layer (done)
The package's public contract, reshaped for the Wave 3 adapter. 101 tests / 0 fail after this The package's public contract, reshaped for the Wave 3 adapter. 101 tests / 0 fail after this
wave; both packages typecheck clean. wave; both packages typecheck clean.
- **`Tool.make` schema flexibility** (`src/tool.ts`): `input`/`output` each accept an Effect - **`Tool.make` schema flexibility** (`src/tool.ts`): `input`/`output` each accept an Effect
Schema (validating, decoded both directions as before) OR a raw JSON Schema document Schema (validating, decoded both directions as before) OR a raw JSON Schema document
(render-only — no validation, values pass through; rendering handles `$defs`/`definitions` (render-only — no validation, values pass through; rendering handles `$defs`/`definitions`
+ `$ref`). `output` is **optional** → signature renders `Promise<unknown>` and the host - `$ref`). `output` is **optional** → signature renders `Promise<unknown>` and the host
result is exposed as-is. Discrimination via `Schema.isSchema`. New helpers exported from result is exposed as-is. Discrimination via `Schema.isSchema`. New helpers exported from
`tool.ts`: `inputTypeScript`/`outputTypeScript`/`decodeInput`/`decodeOutput`/ `tool.ts`: `inputTypeScript`/`outputTypeScript`/`decodeInput`/`decodeOutput`/
`jsonSchemaToTypeScript`; `tool-runtime.ts` consumes them (no direct `Schema.*` use there `jsonSchemaToTypeScript`; `tool-runtime.ts` consumes them (no direct `Schema.*` use there
@@ -235,7 +245,7 @@ wave; both packages typecheck clean.
a final `Effect.map` over every result path (success/timeout/normalized failure). Oversized a final `Effect.map` over every result path (success/timeout/normalized failure). Oversized
serialized values become truncated text + ` [result truncated: N bytes exceeds the M-byte serialized values become truncated text + ` [result truncated: N bytes exceeds the M-byte
output limit; return a smaller value]`; logs keep leading lines within the remaining budget output limit; return a smaller value]`; logs keep leading lines within the remaining budget
+ `[logs truncated: showing K of N lines]`; result gains `truncated: true` (also added to - `[logs truncated: showing K of N lines]`; result gains `truncated: true` (also added to
`ExecuteResultSchema`). UTF-8-safe truncation (no split code points). (The in-sandbox `ExecuteResultSchema`). UTF-8-safe truncation (no split code points). (The in-sandbox
`maxDataBytes` check that used to throw first on oversized raw values died in Fix 5 — `maxDataBytes` check that used to throw first on oversized raw values died in Fix 5 —
truncation is now the only result-size mechanism.) truncation is now the only result-size mechanism.)
@@ -244,6 +254,7 @@ wave; both packages typecheck clean.
(`total: 1`), bypassing ranking. Tokenization/ranking/shape unchanged. (`total: 1`), bypassing ranking. Tokenization/ranking/shape unchanged.
### Wave 3 — OpenCode MCP adapter (done) ### Wave 3 — OpenCode MCP adapter (done)
`packages/opencode/src/session/code-mode.ts` rewritten as a thin adapter over this package; `packages/opencode/src/session/code-mode.ts` rewritten as a thin adapter over this package;
the vendored rune interpreter is gone. Same `define(mcpTools, mcpDefs, servers)` signature, so the vendored rune interpreter is gone. Same `define(mcpTools, mcpDefs, servers)` signature, so
`tools.ts` gating (flag on + MCP tools exist → single `execute` tool, early-return suppresses `tools.ts` gating (flag on + MCP tools exist → single `execute` tool, early-return suppresses
@@ -296,6 +307,7 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
describe/`renderType`/`rankTools` tests died with the old design (58+17+24 → 34+16). describe/`renderType`/`rankTools` tests died with the old design (58+17+24 → 34+16).
### Wave 4 — instructions/prompting + polish (done) ### Wave 4 — instructions/prompting + polish (done)
Instructions are now the budgeted-catalog + prompting-guidance form; verified e2e against a Instructions are now the budgeted-catalog + prompting-guidance form; verified e2e against a
real MCP config. Package still 101 tests / 0 fail; opencode adapter suites still 34 + 16; both real MCP config. Package still 101 tests / 0 fail; opencode adapter suites still 34 + 16; both
packages typecheck clean. packages typecheck clean.
@@ -321,7 +333,7 @@ packages typecheck clean.
exported); `CodeMode.execute` (one-shot) passes it too, preserving the exported); `CodeMode.execute` (one-shot) passes it too, preserving the
`execute`≡`make().execute` law. A speculative `tools.$codemode.search` call on a small `execute`≡`make().execute` law. A speculative `tools.$codemode.search` call on a small
catalog now succeeds instead of `UnknownTool`, and unknown-tool suggestions always point at catalog now succeeds instead of `UnknownTool`, and unknown-tool suggestions always point at
search. Search is *advertised* in the instructions only when the inlined list is PARTIAL, search. Search is _advertised_ in the instructions only when the inlined list is PARTIAL,
keeping small-catalog instructions tight. keeping small-catalog instructions tight.
- **Prompting content** in `instructions()`, mapping 1:1 to the §5 transcript failures: - **Prompting content** in `instructions()`, mapping 1:1 to the §5 transcript failures:
parse-string-results-as-JSON, return-small, console-for-intermediates, and parse-string-results-as-JSON, return-small, console-for-intermediates, and
@@ -352,6 +364,7 @@ packages typecheck clean.
images, output truncation. images, output truncation.
### Wave 5 — Promise generalization (done) ### Wave 5 — Promise generalization (done)
First-class promise values in the interpreter; the direct-tool-call-only `Promise.all` First-class promise values in the interpreter; the direct-tool-call-only `Promise.all`
restriction (and its bespoke AST checks) is gone. Package suite is 136 tests / 0 fail (35 new restriction (and its bespoke AST checks) is gone. Package suite is 136 tests / 0 fail (35 new
in `test/promise.test.ts`); adapter suites and both typechecks unchanged/green; the opencode in `test/promise.test.ts`); adapter suites and both typechecks unchanged/green; the opencode
@@ -526,12 +539,13 @@ adapter needed **no changes**.
**Fix 4 — token-budgeted catalog (was bytes)** (user direction: signatures need a token **Fix 4 — token-budgeted catalog (was bytes)** (user direction: signatures need a token
budget; namespaces must always be present): budget; namespaces must always be present):
- `src/token.ts` added: copy of `@opencode-ai/core/util/token` (`round(chars / 4)`), so - `src/token.ts` added: copy of `@opencode-ai/core/util/token` (`round(chars / 4)`), so
the package stays dependency-free; keep in sync if the core heuristic changes. the package stays dependency-free; keep in sync if the core heuristic changes.
- `DiscoveryOptions.maxInlineCatalogBytes` → `maxInlineCatalogTokens` (default 4,000 - `DiscoveryOptions.maxInlineCatalogBytes` → `maxInlineCatalogTokens` (default 4,000
estimated tokens ≈ the old 16,000 bytes at 4 chars/token — behavior parity, not a size estimated tokens ≈ the old 16,000 bytes at 4 chars/token — behavior parity, not a size
reduction). `discoveryPlan` charges `estimate(catalogLine(tool))` per line; cheapest-first reduction). `discoveryPlan` charges `estimate(catalogLine(tool))` per line; cheapest-first
+ stop-on-first-miss unchanged at the time (stop-on-first-miss replaced by round-robin in - stop-on-first-miss unchanged at the time (stop-on-first-miss replaced by round-robin in
Fix 8). Namespace stub lines were and remain unbudgeted — every Fix 8). Namespace stub lines were and remain unbudgeted — every
namespace always appears with its tool count, even at budget 0 (asserted in package and namespace always appears with its tool count, even at budget 0 (asserted in package and
adapter tests). adapter tests).
@@ -543,6 +557,7 @@ budget; namespaces must always be present):
**Fix 5 — internal limits removed** (user direction: only the three PUBLIC limits survive as **Fix 5 — internal limits removed** (user direction: only the three PUBLIC limits survive as
configurable knobs; the internal limit system dies): configurable knobs; the internal limit system dies):
- `ExecutionLimits` (`timeoutMs` 10_000 / `maxToolCalls` 100 / `maxOutputBytes` 32_000 at - `ExecutionLimits` (`timeoutMs` 10_000 / `maxToolCalls` 100 / `maxOutputBytes` 32_000 at
the time; Fix 6 later removed the first two defaults. Same validation: safe integers, the time; Fix 6 later removed the first two defaults. Same validation: safe integers,
timeoutMs >= 1, others >= 0, RangeError otherwise) is now timeoutMs >= 1, others >= 0, RangeError otherwise) is now
@@ -639,6 +654,7 @@ adapter suites: 34 + 16.
**Fix 8 — condensed instructions + round-robin catalog fairness + plural-aware search** **Fix 8 — condensed instructions + round-robin catalog fairness + plural-aware search**
(user direction: the fixed instruction prose was too verbose; two discovery fixes ride (user direction: the fixed instruction prose was too verbose; two discovery fixes ride
along). All in `tool-runtime.ts`; no interpreter changes. along). All in `tool-runtime.ts`; no interpreter changes.
- **Syntax section inverted**: the three dense allowlist lines (~453 estimated tokens) - **Syntax section inverted**: the three dense allowlist lines (~453 estimated tokens)
are replaced by four short lines (~188) built on "models already know JavaScript; name are replaced by four short lines (~188) built on "models already know JavaScript; name
only what is unusual or missing": (1) standard modern JS works — functions/closures, only what is unusual or missing": (1) standard modern JS works — functions/closures,
@@ -660,7 +676,7 @@ along). All in `tool-runtime.ts`; no interpreter changes.
return-small content now lives ONLY in the numbered Workflow steps (with their return-small content now lives ONLY in the numbered Workflow steps (with their
compliance-driving justifications inline: "most tools return JSON as a string", "raw compliance-driving justifications inline: "most tools return JSON as a string", "raw
payloads get truncated and waste context"); Rules keeps only bullets adding new payloads get truncated and waste context"); Rules keeps only bullets adding new
content — filter/aggregate collections in code, console.* intermediates (logs ride content — filter/aggregate collections in code, console.\* intermediates (logs ride
back), Promise.all parallelism, Object.keys/for...in enumeration, browse-namespace back), Promise.all parallelism, Object.keys/for...in enumeration, browse-namespace
(PARTIAL only), and the media rule compressed to one line. The no-.then/.catch (PARTIAL only), and the media rule compressed to one line. The no-.then/.catch
guidance moved to the Syntax not-supported line. Content upgrades: the PARTIAL search guidance moved to the Syntax not-supported line. Content upgrades: the PARTIAL search
@@ -707,6 +723,7 @@ along). All in `tool-runtime.ts`; no interpreter changes.
**Fix 9 — prompting trims per user review of Fix 8** (user reviewed the condensed **Fix 9 — prompting trims per user review of Fix 8** (user reviewed the condensed
instructions and directed further cuts): instructions and directed further cuts):
- Default `maxInlineCatalogTokens` 4,000 → **2,000** (user wants ~2k tokens of signatures - Default `maxInlineCatalogTokens` 4,000 → **2,000** (user wants ~2k tokens of signatures
auto-inlined; round-robin fairness from Fix 8 spreads it across all namespaces). auto-inlined; round-robin fairness from Fix 8 spreads it across all namespaces).
- Console rule and files/images rule DROPPED from `## Rules`. Replaced by a single - Console rule and files/images rule DROPPED from `## Rules`. Replaced by a single
@@ -725,6 +742,7 @@ instructions and directed further cuts):
**DSL-expansion pass — interpreter-surface batch from §4** (the deferred medium-tier JS **DSL-expansion pass — interpreter-surface batch from §4** (the deferred medium-tier JS
parity items, done as one focused pass; no public API or limit changes): parity items, done as one focused pass; no public API or limit changes):
- **`instanceof` + real Error values**: the `errorConstructors` names (`Error`, - **`instanceof` + real Error values**: the `errorConstructors` names (`Error`,
`TypeError`, `RangeError`, `SyntaxError`, `ReferenceError`, `EvalError`, `URIError`) are `TypeError`, `RangeError`, `SyntaxError`, `ReferenceError`, `EvalError`, `URIError`) are
bound globals (`ErrorConstructorReference`, callable with or without `new`; `typeof` → bound globals (`ErrorConstructorReference`, callable with or without `new`; `typeof` →
@@ -812,6 +830,7 @@ parity items, done as one focused pass; no public API or limit changes):
**Truncation layering — CodeMode truncation off in OpenCode** (user direction; resolves the **Truncation layering — CodeMode truncation off in OpenCode** (user direction; resolves the
§4 outer-truncation item the OPPOSITE way from "kill the outer one"): §4 outer-truncation item the OPPOSITE way from "kill the outer one"):
- `maxOutputBytes` lost its 32,000 default and now behaves exactly like the other two - `maxOutputBytes` lost its 32,000 default and now behaves exactly like the other two
limits: absent = no truncation. All three limits are uniformly no-default — budgets are limits: absent = no truncation. All three limits are uniformly no-default — budgets are
host policy. `ResolvedExecutionLimits.maxOutputBytes` is `number | undefined`; host policy. `ResolvedExecutionLimits.maxOutputBytes` is `number | undefined`;
@@ -837,6 +856,7 @@ regular dependency; hosts depend on it themselves because the API surface is Eff
**Registry promotion + permission-aware catalog** (the "promote to a proper tool service" **Registry promotion + permission-aware catalog** (the "promote to a proper tool service"
restructure; fixes the §4 permission-advertising bug): restructure; fixes the §4 permission-advertising bug):
- **The adapter moved** `src/session/code-mode.ts` → `src/tool/code-mode.ts` and is now a - **The adapter moved** `src/session/code-mode.ts` → `src/tool/code-mode.ts` and is now a
registry-resident tool service on the TaskTool precedent: `CodeModeTool = registry-resident tool service on the TaskTool precedent: `CodeModeTool =
Tool.define(CODE_MODE_TOOL, ...)` whose init depends on `MCP.Service`, `Agent.Service`, Tool.define(CODE_MODE_TOOL, ...)` whose init depends on `MCP.Service`, `Agent.Service`,
@@ -886,6 +906,7 @@ restructure; fixes the §4 permission-advertising bug):
**Shared MCP invocation middle (`McpInvoke.invoke`)** (closes the §4 "plugin hooks skip **Shared MCP invocation middle (`McpInvoke.invoke`)** (closes the §4 "plugin hooks skip
child calls" gap): child calls" gap):
- `packages/opencode/src/mcp/invoke.ts` extracts the duplicated "invoke an MCP tool" - `packages/opencode/src/mcp/invoke.ts` extracts the duplicated "invoke an MCP tool"
middle into one shared `McpInvoke.invoke(input)`: plugin `tool.execute.before` hook → middle into one shared `McpInvoke.invoke(input)`: plugin `tool.execute.before` hook →
permission ask (`{ permission: key, patterns: ["*"], always: ["*"] }` via the caller's permission ask (`{ permission: key, patterns: ["*"], always: ["*"] }` via the caller's
@@ -927,6 +948,7 @@ child calls" gap):
**Signature rendering + compound-assignment parity fixes** (externally reported, both **Signature rendering + compound-assignment parity fixes** (externally reported, both
verified real with failing tests before fixing): verified real with failing tests before fixing):
- **Non-identifier property names in rendered signatures** (`src/tool.ts`): `renderSchema` - **Non-identifier property names in rendered signatures** (`src/tool.ts`): `renderSchema`
emitted raw property names, so schema properties like `foo-bar`/`@type`/`x.y`/`123` emitted raw property names, so schema properties like `foo-bar`/`@type`/`x.y`/`123`
rendered invalid TypeScript (`{ foo-bar?: string }`). Fixed with a `renderKey` helper — rendered invalid TypeScript (`{ foo-bar?: string }`). Fixed with a `renderKey` helper —
@@ -961,8 +983,10 @@ verified real with failing tests before fixing):
## 4. Remaining work (detailed TODO) ## 4. Remaining work (detailed TODO)
### Next DSL-expansion pass (done — see the DSL-expansion pass entry in §3) ### Next DSL-expansion pass (done — see the DSL-expansion pass entry in §3)
Batch these together — per user direction: important, but deliberately deferred to one Batch these together — per user direction: important, but deliberately deferred to one
focused interpreter-surface pass rather than picked off piecemeal. focused interpreter-surface pass rather than picked off piecemeal.
- [x] Medium-tier JS parity items deferred from the original audit: caught errors are plain - [x] Medium-tier JS parity items deferred from the original audit: caught errors are plain
`{ name, message }` objects, not `instanceof Error` (and `Error` isn't a value — `{ name, message }` objects, not `instanceof Error` (and `Error` isn't a value —
`x instanceof Error` is unsupported syntax); `splice` (still a `x instanceof Error` is unsupported syntax); `splice` (still a
@@ -981,6 +1005,7 @@ focused interpreter-surface pass rather than picked off piecemeal.
(`console.log({ m: map })`) — could deep-format instead. (`console.log({ m: map })`) — could deep-format instead.
### Next iteration: text-result handling (deliberate follow-up, user-directed) ### Next iteration: text-result handling (deliberate follow-up, user-directed)
- [ ] Revisit how MCP text results reach the program. Today: `structuredContent` when the - [ ] Revisit how MCP text results reach the program. Today: `structuredContent` when the
server sends it, else joined text as a plain string (the program JSON.parses it, server sends it, else joined text as a plain string (the program JSON.parses it,
guided by a workflow step). Considered and deferred: (a) conservative boundary guided by a workflow step). Considered and deferred: (a) conservative boundary
@@ -992,6 +1017,7 @@ focused interpreter-surface pass rather than picked off piecemeal.
revisit once real usage shows which failure modes matter. revisit once real usage shows which failure modes matter.
### Next iteration: stdlib surface (prioritized) ### Next iteration: stdlib surface (prioritized)
Current instructions say "usual Array/String/Object/Math/JSON methods," but the interpreter is Current instructions say "usual Array/String/Object/Math/JSON methods," but the interpreter is
intentionally a subset. Keep CodeMode focused on orchestration and data shaping, not a full host intentionally a subset. Keep CodeMode focused on orchestration and data shaping, not a full host
runtime, but close the high-friction gaps models are likely to reach for. runtime, but close the high-friction gaps models are likely to reach for.
@@ -1028,7 +1054,9 @@ Explicit non-goals for now: `structuredClone`, `WeakMap`/`WeakSet`, and timers
orchestration use case. orchestration use case.
### Wiring-review findings (subagent code review of the OpenCode integration, triaged) ### Wiring-review findings (subagent code review of the OpenCode integration, triaged)
Pre-PR fixes (user-approved cut): Pre-PR fixes (user-approved cut):
- [x] **Cancellation does not interrupt the interpreter** — the no-limits rationale claimed - [x] **Cancellation does not interrupt the interpreter** — the no-limits rationale claimed
"user cancel interrupts the execution fiber," but `tools.ts` runs tools via "user cancel interrupts the execution fiber," but `tools.ts` runs tools via
`run.promise` → `Effect.runPromise` (`effect/bridge.ts:64-66`) with NO abort wiring; `run.promise` → `Effect.runPromise` (`effect/bridge.ts:64-66`) with NO abort wiring;
@@ -1073,6 +1101,7 @@ Pre-PR fixes (user-approved cut):
`title: "execute"` sites in `code-mode.ts` now reference `CODE_MODE_TOOL`. `title: "execute"` sites in `code-mode.ts` now reference `CODE_MODE_TOOL`.
Post-MVP (logged, not blocking an experimental flag): Post-MVP (logged, not blocking an experimental flag):
- [x] **Plugin `tool.execute.before/after` hooks skip child calls** — legacy MCP - [x] **Plugin `tool.execute.before/after` hooks skip child calls** — legacy MCP
registration fires them per tool (`tools.ts:419-441`); under code mode only the registration fires them per tool (`tools.ts:419-441`); under code mode only the
outer `execute` fires them, so auditing/intercepting plugins silently lose MCP outer `execute` fires them, so auditing/intercepting plugins silently lose MCP
@@ -1105,6 +1134,7 @@ Post-MVP (logged, not blocking an experimental flag):
names that are no longer directly callable under code mode. names that are no longer directly callable under code mode.
### Backlog / loose ends (non-blocking, any order) ### Backlog / loose ends (non-blocking, any order)
- [ ] `evaluateUpdateExpression` (`++`/`--`) still uses raw `Number(current)`, so `d++` on a - [ ] `evaluateUpdateExpression` (`++`/`--`) still uses raw `Number(current)`, so `d++` on a
sandbox Date yields `NaN` where `d += 1` now uses epoch semantics (and real JS `d++` sandbox Date yields `NaN` where `d += 1` now uses epoch semantics (and real JS `d++`
would give epoch+0 numeric). Pre-existing, out of scope of the compound-assignment would give epoch+0 numeric). Pre-existing, out of scope of the compound-assignment
@@ -1141,7 +1171,7 @@ Post-MVP (logged, not blocking an experimental flag):
## 5. Context and gotchas for whoever picks this up ## 5. Context and gotchas for whoever picks this up
- **Motivating failure (why forgiving semantics + prompting matter):** in a real transcript, - **Motivating failure (why forgiving semantics + prompting matter):** in a real transcript,
the model wrote `me.result?.login ?? me.result` where the tool result was a JSON *string* the model wrote `me.result?.login ?? me.result` where the tool result was a JSON _string_
the old strict interpreter threw (`String property 'login' is not available`); then the the old strict interpreter threw (`String property 'login' is not available`); then the
model returned a raw 105KB payload, which native truncation dumped to a file, costing a model returned a raw 105KB payload, which native truncation dumped to a file, costing a
subagent round-trip to extract one number. Interpreter forgiveness stops the crashes; subagent round-trip to extract one number. Interpreter forgiveness stops the crashes;
File diff suppressed because it is too large Load Diff
+1 -7
View File
@@ -1,10 +1,4 @@
export { export { ToolError, CodeMode, ExecuteInputSchema, ExecuteResultSchema, toolError } from "./codemode.js"
ToolError,
CodeMode,
ExecuteInputSchema,
ExecuteResultSchema,
toolError,
} from "./codemode.js"
export { Tool } from "./tool.js" export { Tool } from "./tool.js"
export type { Definition as ToolDefinition, JsonSchema, ToolSchema } from "./tool.js" export type { Definition as ToolDefinition, JsonSchema, ToolSchema } from "./tool.js"
export type { ToolCallEnded, ToolCallHooks } from "./tool-runtime.js" export type { ToolCallEnded, ToolCallHooks } from "./tool-runtime.js"
+131 -48
View File
@@ -21,10 +21,15 @@ export type HostTools<R = never> = {
export type Services<Tools> = Tools extends (...args: Array<unknown>) => Effect.Effect<unknown, unknown, infer R> export type Services<Tools> = Tools extends (...args: Array<unknown>) => Effect.Effect<unknown, unknown, infer R>
? R ? R
: Tools extends { readonly _tag: "CodeModeTool"; readonly run: (input: unknown) => Effect.Effect<unknown, unknown, infer R> } : Tools extends {
readonly _tag: "CodeModeTool"
readonly run: (input: unknown) => Effect.Effect<unknown, unknown, infer R>
}
? R ? R
: Tools extends object : Tools extends object
? string extends keyof Tools ? never : Services<Tools[keyof Tools]> ? string extends keyof Tools
? never
: Services<Tools[keyof Tools]>
: never : never
/** Minimal audit record retained for each admitted tool call. */ /** Minimal audit record retained for each admitted tool call. */
@@ -68,11 +73,13 @@ export type SafeObject = Record<string, unknown>
const reservedNamespace = "$codemode" const reservedNamespace = "$codemode"
const defaultMaxInlineCatalogTokens = 2_000 const defaultMaxInlineCatalogTokens = 2_000
const defaultSearchLimit = 10 const defaultSearchLimit = 10
const searchSignature = "tools.$codemode.search({ query?: string, namespace?: string, limit?: number }): Promise<{ items: Array<{ path: string; description: string; signature: string }>; total: number }>" const searchSignature =
"tools.$codemode.search({ query?: string, namespace?: string, limit?: number }): Promise<{ items: Array<{ path: string; description: string; signature: string }>; total: number }>"
const toolExpression = (path: string) => const toolExpression = (path: string) =>
"tools" + path "tools" +
path
.split(".") .split(".")
.map((segment) => identifierSegment.test(segment) ? `.${segment}` : `[${JSON.stringify(segment)}]`) .map((segment) => (identifierSegment.test(segment) ? `.${segment}` : `[${JSON.stringify(segment)}]`))
.join("") .join("")
export class ToolReference { export class ToolReference {
@@ -88,7 +95,12 @@ const MAX_VALUE_DEPTH = 32
export class ToolRuntimeError extends Error { export class ToolRuntimeError extends Error {
constructor( constructor(
readonly kind: "UnknownTool" | "InvalidToolInput" | "InvalidToolOutput" | "InvalidDataValue" | "ToolCallLimitExceeded", readonly kind:
| "UnknownTool"
| "InvalidToolInput"
| "InvalidToolOutput"
| "InvalidDataValue"
| "ToolCallLimitExceeded",
message: string, message: string,
readonly suggestions: ReadonlyArray<string> = [], readonly suggestions: ReadonlyArray<string> = [],
) { ) {
@@ -131,7 +143,13 @@ export const isBlockedMember = (name: string): boolean => blockedMemberNames.has
export const copyIn = (value: unknown, label: string, preserveSandboxValues = false): unknown => export const copyIn = (value: unknown, label: string, preserveSandboxValues = false): unknown =>
copyBounded(value, label, 0, new Set(), preserveSandboxValues) copyBounded(value, label, 0, new Set(), preserveSandboxValues)
const copyBounded = (value: unknown, label: string, depth: number, seen: Set<object>, preserveSandboxValues: boolean): unknown => { const copyBounded = (
value: unknown,
label: string,
depth: number,
seen: Set<object>,
preserveSandboxValues: boolean,
): unknown => {
if (depth > MAX_VALUE_DEPTH) { if (depth > MAX_VALUE_DEPTH) {
throw new ToolRuntimeError("InvalidDataValue", `${label} exceeds the maximum value depth of ${MAX_VALUE_DEPTH}.`) throw new ToolRuntimeError("InvalidDataValue", `${label} exceeds the maximum value depth of ${MAX_VALUE_DEPTH}.`)
} }
@@ -166,7 +184,12 @@ const copyBounded = (value: unknown, label: string, depth: number, seen: Set<obj
// Intra-sandbox checkpoints keep sandbox value instances alive as leaves; their contents // Intra-sandbox checkpoints keep sandbox value instances alive as leaves; their contents
// are never walked here (Map/Set members are validated where mutation happens, and the // are never walked here (Map/Set members are validated where mutation happens, and the
// real boundary still serializes them below). // real boundary still serializes them below).
if (value instanceof SandboxDate || value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet) { if (
value instanceof SandboxDate ||
value instanceof SandboxRegExp ||
value instanceof SandboxMap ||
value instanceof SandboxSet
) {
return value return value
} }
// Host instances cannot normally reach an intra-sandbox checkpoint (tool results cross // Host instances cannot normally reach an intra-sandbox checkpoint (tool results cross
@@ -197,8 +220,12 @@ const copyBounded = (value: unknown, label: string, depth: number, seen: Set<obj
return Number.isFinite(value.getTime()) ? value.toISOString() : null return Number.isFinite(value.getTime()) ? value.toISOString() : null
} }
if ( if (
value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet || value instanceof SandboxRegExp ||
value instanceof RegExp || value instanceof Map || value instanceof Set value instanceof SandboxMap ||
value instanceof SandboxSet ||
value instanceof RegExp ||
value instanceof Map ||
value instanceof Set
) { ) {
return Object.create(null) as SafeObject return Object.create(null) as SafeObject
} }
@@ -250,7 +277,10 @@ export const copyOut = (value: unknown, undefinedAsNull = false): unknown => {
return value return value
} }
const definitions = <R>(tools: HostTools<R>, path: ReadonlyArray<string> = []): Array<{ path: string; definition: Definition<R> }> => { const definitions = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string> = [],
): Array<{ path: string; definition: Definition<R> }> => {
const entries: Array<{ path: string; definition: Definition<R> }> = [] const entries: Array<{ path: string; definition: Definition<R> }> = []
for (const [name, value] of Object.entries(tools)) { for (const [name, value] of Object.entries(tools)) {
const next = [...path, name] const next = [...path, name]
@@ -342,8 +372,11 @@ const toSearchEntry = <R>(path: string, definition: Definition<R>, description:
path, path,
definition.description, definition.description,
...inputProperties(definition).flatMap(({ name, description: property }) => ...inputProperties(definition).flatMap(({ name, description: property }) =>
property === undefined ? [name] : [name, property]), property === undefined ? [name] : [name, property],
].join("\n").toLowerCase(), ),
]
.join("\n")
.toLowerCase(),
}) })
/** The runtime search index over every described tool. Search is always registered. */ /** The runtime search index over every described tool. Search is always registered. */
@@ -396,7 +429,8 @@ export const discoveryPlan = <R>(
namespace, namespace,
picked: new Set<ToolDescription>(), picked: new Set<ToolDescription>(),
queue: [...group].sort( queue: [...group].sort(
(left, right) => estimate(catalogLine(left)) - estimate(catalogLine(right)) || left.path.localeCompare(right.path), (left, right) =>
estimate(catalogLine(left)) - estimate(catalogLine(right)) || left.path.localeCompare(right.path),
), ),
})) }))
let used = 0 let used = 0
@@ -472,7 +506,9 @@ export const discoveryPlan = <R>(
"- A result typed `Promise<unknown>` has no guaranteed shape — verify what actually came back before relying on its fields.", "- A result typed `Promise<unknown>` has no guaranteed shape — verify what actually came back before relying on its fields.",
"- Run independent calls in parallel: `await Promise.all(items.map((item) => tools.<namespace>.<tool>(item)))`.", "- Run independent calls in parallel: `await Promise.all(items.map((item) => tools.<namespace>.<tool>(item)))`.",
"- `Object.keys(tools)` lists namespaces; `Object.keys(tools.<namespace>)` lists its tools; `for...in` works on both.", "- `Object.keys(tools)` lists namespaces; `Object.keys(tools.<namespace>)` lists its tools; `for...in` works on both.",
...(complete ? [] : ['- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.']), ...(complete
? []
: ['- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.']),
] ]
const syntax = [ const syntax = [
@@ -500,26 +536,21 @@ export const discoveryPlan = <R>(
const count = `${group.length} tool${group.length === 1 ? "" : "s"}` const count = `${group.length} tool${group.length === 1 ? "" : "s"}`
// Annotate only when a namespace is not fully shown, so a comprehensive // Annotate only when a namespace is not fully shown, so a comprehensive
// namespace reads cleanly and a truncated one is unambiguous. // namespace reads cleanly and a truncated one is unambiguous.
const label = picked.size === group.length ? count : picked.size === 0 ? `${count}, none shown` : `${count}, ${picked.size} shown` const label =
picked.size === group.length
? count
: picked.size === 0
? `${count}, none shown`
: `${count}, ${picked.size} shown`
toolSection.push(`- ${namespace} (${label})`) toolSection.push(`- ${namespace} (${label})`)
for (const tool of group) if (picked.has(tool)) toolSection.push(catalogLine(tool)) for (const tool of group) if (picked.has(tool)) toolSection.push(catalogLine(tool))
} }
if (!complete) { if (!complete) {
toolSection.push( toolSection.push("", "Search returns complete callable signatures:", `- ${searchSignature}`)
"",
"Search returns complete callable signatures:",
`- ${searchSignature}`,
)
} }
} }
const lines = [ const lines = [...intro, ...workflow, ...rules, ...syntax, ...toolSection]
...intro,
...workflow,
...rules,
...syntax,
...toolSection,
]
return { return {
catalog: described, catalog: described,
instructions: lines.join("\n"), instructions: lines.join("\n"),
@@ -534,18 +565,29 @@ export const discoveryPlan = <R>(
* function in JS). An unknown path is an `UnknownTool` error pointing at the working * function in JS). An unknown path is an `UnknownTool` error pointing at the working
* discovery idioms, mirroring how calling an unknown tool fails. * discovery idioms, mirroring how calling an unknown tool fails.
*/ */
const namespaceKeys = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, searchEnabled: boolean): ReadonlyArray<string> => { const namespaceKeys = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string>,
searchEnabled: boolean,
): ReadonlyArray<string> => {
// The reserved discovery namespace is virtual (never present in the host tree); enumerate // The reserved discovery namespace is virtual (never present in the host tree); enumerate
// it explicitly so `Object.keys(tools.$codemode)` matches the callable surface. // it explicitly so `Object.keys(tools.$codemode)` matches the callable surface.
if (searchEnabled && path.length === 1 && path[0] === reservedNamespace) return ["search"] if (searchEnabled && path.length === 1 && path[0] === reservedNamespace) return ["search"]
let value: HostTool<R> | Definition<R> | HostTools<R> = tools let value: HostTool<R> | Definition<R> | HostTools<R> = tools
for (const segment of path) { for (const segment of path) {
if (isBlockedMember(segment) || typeof value === "function" || isDefinition(value) || !Object.hasOwn(value, segment)) { if (
isBlockedMember(segment) ||
typeof value === "function" ||
isDefinition(value) ||
!Object.hasOwn(value, segment)
) {
throw new ToolRuntimeError( throw new ToolRuntimeError(
"UnknownTool", "UnknownTool",
`Unknown tool namespace '${path.join(".")}'.`, `Unknown tool namespace '${path.join(".")}'.`,
searchEnabled searchEnabled
? ["Object.keys(tools) lists the available namespaces; tools.$codemode.search({ query }) finds described tools."] ? [
"Object.keys(tools) lists the available namespaces; tools.$codemode.search({ query }) finds described tools.",
]
: ["Object.keys(tools) lists the available namespaces."], : ["Object.keys(tools) lists the available namespaces."],
) )
} }
@@ -555,12 +597,25 @@ const namespaceKeys = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, sear
return Object.keys(value) return Object.keys(value)
} }
const resolve = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, searchEnabled: boolean): HostTool<R> | Definition<R> => { const resolve = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string>,
searchEnabled: boolean,
): HostTool<R> | Definition<R> => {
let value: HostTool<R> | Definition<R> | HostTools<R> = tools let value: HostTool<R> | Definition<R> | HostTools<R> = tools
for (const segment of path) { for (const segment of path) {
if (isBlockedMember(segment) || typeof value === "function" || isDefinition(value) || !Object.hasOwn(value, segment)) { if (
throw new ToolRuntimeError("UnknownTool", `Unknown tool '${path.join(".")}'.`, searchEnabled ? ["Use tools.$codemode.search({ query }) to find available described tools."] : []) isBlockedMember(segment) ||
typeof value === "function" ||
isDefinition(value) ||
!Object.hasOwn(value, segment)
) {
throw new ToolRuntimeError(
"UnknownTool",
`Unknown tool '${path.join(".")}'.`,
searchEnabled ? ["Use tools.$codemode.search({ query }) to find available described tools."] : [],
)
} }
value = value[segment] as HostTool<R> | Definition<R> | HostTools<R> value = value[segment] as HostTool<R> | Definition<R> | HostTools<R>
} }
@@ -602,7 +657,8 @@ export const make = <R>(
return effect.pipe( return effect.pipe(
Effect.tap(() => onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "success" })), Effect.tap(() => onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "success" })),
Effect.tapError((error) => Effect.tapError((error) =>
onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "failure", message: failureMessage(error) })), onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "failure", message: failureMessage(error) }),
),
) )
} }
@@ -637,17 +693,32 @@ export const make = <R>(
if (!searchEnabled) throw new ToolRuntimeError("UnknownTool", `Unknown tool '${name}'.`) if (!searchEnabled) throw new ToolRuntimeError("UnknownTool", `Unknown tool '${name}'.`)
const input = externalArgs[0] const input = externalArgs[0]
if (externalArgs.length !== 1 || input === null || typeof input !== "object" || Array.isArray(input)) { if (externalArgs.length !== 1 || input === null || typeof input !== "object" || Array.isArray(input)) {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search expects { query?: string; namespace?: string; limit?: number }.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search expects { query?: string; namespace?: string; limit?: number }.",
)
} }
const request = input as { query?: unknown; namespace?: unknown; limit?: unknown } const request = input as { query?: unknown; namespace?: unknown; limit?: unknown }
if (request.query !== undefined && typeof request.query !== "string") { if (request.query !== undefined && typeof request.query !== "string") {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search query must be a string when provided.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search query must be a string when provided.",
)
} }
if (request.namespace !== undefined && typeof request.namespace !== "string") { if (request.namespace !== undefined && typeof request.namespace !== "string") {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search namespace must be a string when provided.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search namespace must be a string when provided.",
)
} }
if (request.limit !== undefined && (typeof request.limit !== "number" || !Number.isSafeInteger(request.limit) || request.limit <= 0)) { if (
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search limit must be a positive safe integer when provided.") request.limit !== undefined &&
(typeof request.limit !== "number" || !Number.isSafeInteger(request.limit) || request.limit <= 0)
) {
throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search limit must be a positive safe integer when provided.",
)
} }
const query = typeof request.query === "string" ? request.query : "" const query = typeof request.query === "string" ? request.query : ""
const namespace = typeof request.namespace === "string" ? request.namespace : undefined const namespace = typeof request.namespace === "string" ? request.namespace : undefined
@@ -656,20 +727,27 @@ export const make = <R>(
Effect.try({ Effect.try({
try: () => { try: () => {
const limit = typeof request.limit === "number" ? request.limit : defaultSearchLimit const limit = typeof request.limit === "number" ? request.limit : defaultSearchLimit
const scoped = namespace === undefined ? searchIndex : searchIndex.filter((entry) => entry.namespace === namespace) const scoped =
namespace === undefined ? searchIndex : searchIndex.filter((entry) => entry.namespace === namespace)
// A query that names one tool path exactly (canonical path or rendered // A query that names one tool path exactly (canonical path or rendered
// JavaScript expression) is a lookup, not a search: return that tool alone. // JavaScript expression) is a lookup, not a search: return that tool alone.
const trimmed = query.trim() const trimmed = query.trim()
const pathQuery = trimmed.startsWith("tools.") ? trimmed.slice("tools.".length) : trimmed const pathQuery = trimmed.startsWith("tools.") ? trimmed.slice("tools.".length) : trimmed
const exact = pathQuery === "" ? undefined : scoped.find((entry) => const exact =
entry.description.path === pathQuery || toolExpression(entry.description.path) === trimmed) pathQuery === ""
? undefined
: scoped.find(
(entry) =>
entry.description.path === pathQuery || toolExpression(entry.description.path) === trimmed,
)
const terms = tokenize(query).map(termForms) const terms = tokenize(query).map(termForms)
// Additive field-weighted scoring, summed across terms: exact path or path // Additive field-weighted scoring, summed across terms: exact path or path
// segment (20) > path substring (8) > description substring (4) > any // segment (20) > path substring (8) > description substring (4) > any
// searchable text, incl. input parameter names/descriptions (2). Each term // searchable text, incl. input parameter names/descriptions (2). Each term
// matches a field when any of its forms (the term or a singular variant) // matches a field when any of its forms (the term or a singular variant)
// does. An empty query browses everything, alphabetical by path. // does. An empty query browses everything, alphabetical by path.
const ranked = exact !== undefined const ranked =
exact !== undefined
? [exact] ? [exact]
: scoped : scoped
.map((entry) => { .map((entry) => {
@@ -687,8 +765,11 @@ export const make = <R>(
return { entry, score } return { entry, score }
}) })
.filter(({ score }) => terms.length === 0 || score > 0) .filter(({ score }) => terms.length === 0 || score > 0)
.sort((left, right) => .sort(
right.score - left.score || left.entry.description.path.localeCompare(right.entry.description.path)) (left, right) =>
right.score - left.score ||
left.entry.description.path.localeCompare(right.entry.description.path),
)
.map(({ entry }) => entry) .map(({ entry }) => entry)
// Result paths are rendered as JavaScript expressions so each `path` is // Result paths are rendered as JavaScript expressions so each `path` is
// directly usable as the call site (`await tools.github.list({ ... })` or // directly usable as the call site (`await tools.github.list({ ... })` or
@@ -711,10 +792,12 @@ export const make = <R>(
const tool = resolve(tools, path, searchEnabled) const tool = resolve(tools, path, searchEnabled)
let describedInput: unknown let describedInput: unknown
if (isDefinition(tool)) { if (isDefinition(tool)) {
if (externalArgs.length !== 1) throw new ToolRuntimeError("InvalidToolInput", `Tool '${name}' expects exactly one input object.`) if (externalArgs.length !== 1)
throw new ToolRuntimeError("InvalidToolInput", `Tool '${name}' expects exactly one input object.`)
describedInput = yield* Effect.try({ describedInput = yield* Effect.try({
try: () => decodeToolInput(tool, externalArgs[0]), try: () => decodeToolInput(tool, externalArgs[0]),
catch: (cause) => new ToolRuntimeError("InvalidToolInput", `Invalid input for tool '${name}': ${String(cause)}`), catch: (cause) =>
new ToolRuntimeError("InvalidToolInput", `Invalid input for tool '${name}': ${String(cause)}`),
}) })
} }
const input = isDefinition(tool) ? describedInput : externalArgs const input = isDefinition(tool) ? describedInput : externalArgs
+27 -12
View File
@@ -57,8 +57,7 @@ export type Options<I extends ToolSchema, O extends ToolSchema | undefined, R =
export const isDefinition = <R = never>(value: unknown): value is Definition<R> => export const isDefinition = <R = never>(value: unknown): value is Definition<R> =>
typeof value === "object" && value !== null && "_tag" in value && value._tag === "CodeModeTool" typeof value === "object" && value !== null && "_tag" in value && value._tag === "CodeModeTool"
const isEffectSchema = (schema: ToolSchema): schema is Schema.Decoder<unknown> & Schema.Top => const isEffectSchema = (schema: ToolSchema): schema is Schema.Decoder<unknown> & Schema.Top => Schema.isSchema(schema)
Schema.isSchema(schema)
const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unknown" const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unknown"
@@ -69,10 +68,12 @@ const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unkn
export const identifierSegment = /^[A-Za-z_$][A-Za-z0-9_$]*$/ export const identifierSegment = /^[A-Za-z_$][A-Za-z0-9_$]*$/
/** Renders a property name as a valid TS object key: bare when an identifier, quoted otherwise. */ /** Renders a property name as a valid TS object key: bare when an identifier, quoted otherwise. */
const renderKey = (name: string): string => identifierSegment.test(name) ? name : JSON.stringify(name) const renderKey = (name: string): string => (identifierSegment.test(name) ? name : JSON.stringify(name))
const effectNumberSentinel = (schema: JsonSchema) => const effectNumberSentinel = (schema: JsonSchema) =>
schema.type === "string" && Array.isArray(schema.enum) && schema.enum.length === 1 && schema.type === "string" &&
Array.isArray(schema.enum) &&
schema.enum.length === 1 &&
(schema.enum[0] === "NaN" || schema.enum[0] === "Infinity" || schema.enum[0] === "-Infinity") (schema.enum[0] === "NaN" || schema.enum[0] === "Infinity" || schema.enum[0] === "-Infinity")
/** /**
@@ -118,7 +119,8 @@ const docTags = (schema: JsonSchema): Array<string> => {
*/ */
const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad: string): string => { const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad: string): string => {
const lines = [...(description === undefined ? [] : description.split("\n")), ...tags].map((line) => const lines = [...(description === undefined ? [] : description.split("\n")), ...tags].map((line) =>
line.replaceAll("*/", "* /").replace(/\s+$/, "")) line.replaceAll("*/", "* /").replace(/\s+$/, ""),
)
while (lines.length > 0 && lines[0]!.trim() === "") lines.shift() while (lines.length > 0 && lines[0]!.trim() === "") lines.shift()
while (lines.length > 0 && lines[lines.length - 1]!.trim() === "") lines.pop() while (lines.length > 0 && lines[lines.length - 1]!.trim() === "") lines.pop()
if (lines.length === 0) return "" if (lines.length === 0) return ""
@@ -127,7 +129,12 @@ const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad
return `${pad}/**\n${body}\n${pad} */\n` return `${pad}/**\n${body}\n${pad} */\n`
} }
const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: ReadonlySet<string> = new Set()): string => { const renderSchema = (
schema: JsonSchema,
ctx: RenderContext,
depth = 0,
seen: ReadonlySet<string> = new Set(),
): string => {
if (depth > MAX_RENDER_DEPTH) return "unknown" if (depth > MAX_RENDER_DEPTH) return "unknown"
if (schema.$ref) { if (schema.$ref) {
const name = schema.$ref.split("/").pop() const name = schema.$ref.split("/").pop()
@@ -146,13 +153,16 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
if ( if (
alternatives.some((item) => item.type === "number") && alternatives.some((item) => item.type === "number") &&
alternatives.every((item) => item.type === "number" || effectNumberSentinel(item)) alternatives.every((item) => item.type === "number" || effectNumberSentinel(item))
) return "number" )
return "number"
// An empty Schema.Struct({}) emits `anyOf: [{ type: "object" }, { type: "array" }]` // An empty Schema.Struct({}) emits `anyOf: [{ type: "object" }, { type: "array" }]`
// (no properties/items); render the bare shape as {} instead of `{} | Array<unknown>`. // (no properties/items); render the bare shape as {} instead of `{} | Array<unknown>`.
if ( if (
alternatives.length === 2 && alternatives.length === 2 &&
alternatives[0]?.type === "object" && alternatives[0].properties === undefined && alternatives[0]?.type === "object" &&
alternatives[1]?.type === "array" && alternatives[1].items === undefined alternatives[0].properties === undefined &&
alternatives[1]?.type === "array" &&
alternatives[1].items === undefined
) { ) {
return "{}" return "{}"
} }
@@ -170,7 +180,8 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
const required = new Set(schema.required ?? []) const required = new Set(schema.required ?? [])
const properties = Object.entries(schema.properties ?? {}) const properties = Object.entries(schema.properties ?? {})
const additional = schema.additionalProperties const additional = schema.additionalProperties
const indexType = additional && typeof additional === "object" ? renderSchema(additional, ctx, depth + 1, seen) : undefined const indexType =
additional && typeof additional === "object" ? renderSchema(additional, ctx, depth + 1, seen) : undefined
const field = ([name, value]: readonly [string, JsonSchema]) => const field = ([name, value]: readonly [string, JsonSchema]) =>
`${renderKey(name)}${required.has(name) ? "" : "?"}: ${renderSchema(value, ctx, depth + 1, seen)}` `${renderKey(name)}${required.has(name) ? "" : "?"}: ${renderSchema(value, ctx, depth + 1, seen)}`
@@ -183,7 +194,9 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
// Pretty: an indented block, each described field preceded by its JSDoc comment. // Pretty: an indented block, each described field preceded by its JSDoc comment.
if (properties.length === 0 && indexType === undefined) return "{}" if (properties.length === 0 && indexType === undefined) return "{}"
const pad = " ".repeat(depth + 1) const pad = " ".repeat(depth + 1)
const lines = properties.map((entry) => `${jsdoc(entry[1].description, docTags(entry[1]), pad)}${pad}${field(entry)}`) const lines = properties.map(
(entry) => `${jsdoc(entry[1].description, docTags(entry[1]), pad)}${pad}${field(entry)}`,
)
if (indexType !== undefined) lines.push(`${pad}[key: string]: ${indexType}`) if (indexType !== undefined) lines.push(`${pad}[key: string]: ${indexType}`)
return `{\n${lines.join("\n")}\n${" ".repeat(depth)}}` return `{\n${lines.join("\n")}\n${" ".repeat(depth)}}`
} }
@@ -262,7 +275,9 @@ export const inputProperties = <R>(definition: Definition<R>): Array<InputProper
* fields; the default stays the compact single-line form. * fields; the default stays the compact single-line form.
*/ */
export const inputTypeScript = <R>(definition: Definition<R>, pretty = false): string => export const inputTypeScript = <R>(definition: Definition<R>, pretty = false): string =>
isEffectSchema(definition.input) ? toTypeScript(definition.input, false, pretty) : jsonSchemaToTypeScript(definition.input, pretty) isEffectSchema(definition.input)
? toTypeScript(definition.input, false, pretty)
: jsonSchemaToTypeScript(definition.input, pretty)
/** /**
* The model-visible TypeScript type of a tool's result; tools without an output schema * The model-visible TypeScript type of a tool's result; tools without an output schema
+4 -1
View File
@@ -49,4 +49,7 @@ export class SandboxSet {
} }
export const isSandboxValue = (value: unknown): value is SandboxDate | SandboxRegExp | SandboxMap | SandboxSet => export const isSandboxValue = (value: unknown): value is SandboxDate | SandboxRegExp | SandboxMap | SandboxSet =>
value instanceof SandboxDate || value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet value instanceof SandboxDate ||
value instanceof SandboxRegExp ||
value instanceof SandboxMap ||
value instanceof SandboxSet
+209 -114
View File
@@ -1,6 +1,13 @@
import { describe, expect, test } from "bun:test" import { describe, expect, test } from "bun:test"
import { Cause, Effect, Schema } from "effect" import { Cause, Effect, Schema } from "effect"
import { CodeMode, ExecuteInputSchema, ExecuteResultSchema, Tool, toolError, type ExecutionLimits } from "../src/index.js" import {
CodeMode,
ExecuteInputSchema,
ExecuteResultSchema,
Tool,
toolError,
type ExecutionLimits,
} from "../src/index.js"
import type { Definition } from "../src/tool.js" import type { Definition } from "../src/tool.js"
const run = (tool: Definition<never>) => const run = (tool: Definition<never>) =>
@@ -75,11 +82,14 @@ describe("CodeMode host failure boundary", () => {
output: Schema.Unknown, output: Schema.Unknown,
run: () => run: () =>
Effect.succeed( Effect.succeed(
new Proxy({}, { new Proxy(
{},
{
ownKeys: () => { ownKeys: () => {
throw new Error("host-output-secret") throw new Error("host-output-secret")
}, },
}), },
),
), ),
}), }),
) )
@@ -162,9 +172,7 @@ describe("CodeMode tool-call observation", () => {
) )
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(calls).toStrictEqual([ expect(calls).toStrictEqual([{ index: 0, name: "context.lookup", input: { query: "deployment failure" } }])
{ index: 0, name: "context.lookup", input: { query: "deployment failure" } },
])
}) })
test("observes settled calls with outcome and duration", async () => { test("observes settled calls with outcome and duration", async () => {
@@ -173,16 +181,17 @@ describe("CodeMode tool-call observation", () => {
description: "Look up a value", description: "Look up a value",
input: Schema.Struct({ query: Schema.String }), input: Schema.Struct({ query: Schema.String }),
output: Schema.String, output: Schema.String,
run: ({ query }) => run: ({ query }) => (query === "boom" ? Effect.fail(toolError("Lookup refused")) : Effect.succeed(query)),
query === "boom" ? Effect.fail(toolError("Lookup refused")) : Effect.succeed(query),
}) })
const runtime = CodeMode.make({ const runtime = CodeMode.make({
tools: { context: { lookup } }, tools: { context: { lookup } },
onToolCallStart: (call) => Effect.sync(() => { onToolCallStart: (call) =>
Effect.sync(() => {
events.push({ phase: "start", index: call.index, name: call.name }) events.push({ phase: "start", index: call.index, name: call.name })
}), }),
onToolCallEnd: (call) => Effect.sync(() => { onToolCallEnd: (call) =>
Effect.sync(() => {
expect(call.durationMs).toBeGreaterThanOrEqual(0) expect(call.durationMs).toBeGreaterThanOrEqual(0)
events.push({ events.push({
phase: "end", phase: "end",
@@ -210,13 +219,15 @@ describe("CodeMode tool-call observation", () => {
describe("CodeMode console capture", () => { describe("CodeMode console capture", () => {
test("captures console output as bounded result logs", async () => { test("captures console output as bounded result logs", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
const returned = console.log("Thread info:", { name: "Demo", count: 2 }) const returned = console.log("Thread info:", { name: "Demo", count: 2 })
console.warn("careful") console.warn("careful")
return returned return returned
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
@@ -228,39 +239,45 @@ describe("CodeMode console capture", () => {
}) })
test("keeps logs captured before failures", async () => { test("keeps logs captured before failures", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log("before failure") console.log("before failure")
throw new Error("boom") throw new Error("boom")
`, `,
})) }),
)
expect(result.ok ? undefined : result.logs).toStrictEqual(["before failure"]) expect(result.ok ? undefined : result.logs).toStrictEqual(["before failure"])
expect(result.ok ? undefined : result.error.message).toBe("Uncaught: boom") expect(result.ok ? undefined : result.error.message).toBe("Uncaught: boom")
}) })
test("prints NaN and Infinity literally instead of the JSON null", async () => { test("prints NaN and Infinity literally instead of the JSON null", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log(NaN) console.log(NaN)
console.log(Infinity, -Infinity) console.log(Infinity, -Infinity)
console.log({ ratio: NaN, bounds: [Infinity] }) console.log({ ratio: NaN, bounds: [Infinity] })
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual(["NaN", "Infinity -Infinity", '{"ratio":NaN,"bounds":[Infinity]}']) expect(result.logs).toStrictEqual(["NaN", "Infinity -Infinity", '{"ratio":NaN,"bounds":[Infinity]}'])
}) })
test("renders sandbox values nested inside logged containers", async () => { test("renders sandbox values nested inside logged containers", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log({ m: new Map([["a", 1]]), when: new Date(0), r: /ab/g, s: new Set([1, 2]) }) console.log({ m: new Map([["a", 1]]), when: new Date(0), r: /ab/g, s: new Set([1, 2]) })
console.log([new Date(0)]) console.log([new Date(0)])
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual([ expect(result.logs).toStrictEqual([
@@ -270,7 +287,8 @@ describe("CodeMode console capture", () => {
}) })
test("console formatting is total: cycles and opaque references render as markers", async () => { test("console formatting is total: cycles and opaque references render as markers", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
const m = new Map() const m = new Map()
m.set("self", m) m.set("self", m)
@@ -278,29 +296,30 @@ describe("CodeMode console capture", () => {
console.log({ fn: (x) => x, ok: 1 }) console.log({ fn: (x) => x, ok: 1 })
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual([ expect(result.logs).toStrictEqual(['{"box":Map(1) [["self",[Circular]]]}', '{"fn":[CodeMode reference],"ok":1}'])
'{"box":Map(1) [["self",[Circular]]]}',
'{"fn":[CodeMode reference],"ok":1}',
])
}) })
test("console.table renders sandbox value cells", async () => { test("console.table renders sandbox value cells", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.table([{ when: new Date(0), n: NaN }]) console.table([{ when: new Date(0), n: NaN }])
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual(["(index)\twhen\tn\n0\t1970-01-01T00:00:00.000Z\tNaN"]) expect(result.logs).toStrictEqual(["(index)\twhen\tn\n0\t1970-01-01T00:00:00.000Z\tNaN"])
}) })
test("captures console.dir and console.table output", async () => { test("captures console.dir and console.table output", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.dir({ nested: { ok: true } }) console.dir({ nested: { ok: true } })
console.table([ console.table([
@@ -309,15 +328,13 @@ describe("CodeMode console capture", () => {
], ["name", "count"]) ], ["name", "count"])
return "done" return "done"
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: "done", value: "done",
logs: [ logs: ['{"nested":{"ok":true}}', "(index)\tname\tcount\n0\tKit\t1\n1\tOlive\t2"],
'{"nested":{"ok":true}}',
"(index)\tname\tcount\n0\tKit\t1\n1\tOlive\t2",
],
toolCalls: [], toolCalls: [],
}) })
}) })
@@ -325,9 +342,11 @@ describe("CodeMode console capture", () => {
describe("CodeMode output budget", () => { describe("CodeMode output budget", () => {
test("absent maxOutputBytes means no truncation at all", async () => { test("absent maxOutputBytes means no truncation at all", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: `console.log("z".repeat(50_000)); return "x".repeat(100_000)`, code: `console.log("z".repeat(50_000)); return "x".repeat(100_000)`,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
@@ -338,29 +357,35 @@ describe("CodeMode output budget", () => {
test("truncates an oversized result value with a marker instead of failing", async () => { test("truncates an oversized result value with a marker instead of failing", async () => {
const limits: ExecutionLimits = { maxOutputBytes: 40 } const limits: ExecutionLimits = { maxOutputBytes: 40 }
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: `return { data: "${"x".repeat(200)}" }`, code: `return { data: "${"x".repeat(200)}" }`,
limits, limits,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
expect(result.truncated).toBe(true) expect(result.truncated).toBe(true)
expect(typeof result.value).toBe("string") expect(typeof result.value).toBe("string")
expect(result.value).toMatch(/^\{"data":"x+ \[result truncated: \d+ bytes exceeds the 40-byte output limit; return a smaller value\]$/) expect(result.value).toMatch(
/^\{"data":"x+ \[result truncated: \d+ bytes exceeds the 40-byte output limit; return a smaller value\]$/,
)
expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(result)))).toStrictEqual(result) expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(result)))).toStrictEqual(result)
}) })
test("keeps leading logs within the remaining budget and marks the cut", async () => { test("keeps leading logs within the remaining budget and marks the cut", async () => {
const limits: ExecutionLimits = { maxOutputBytes: 40 } const limits: ExecutionLimits = { maxOutputBytes: 40 }
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log("first line") console.log("first line")
console.log("${"y".repeat(200)}") console.log("${"y".repeat(200)}")
return "ok" return "ok"
`, `,
limits, limits,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
@@ -370,12 +395,14 @@ describe("CodeMode output budget", () => {
}) })
test("does not mark results within the budget", async () => { test("does not mark results within the budget", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log("fits") console.log("fits")
return { fits: true } return { fits: true }
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: { fits: true }, value: { fits: true },
@@ -395,18 +422,21 @@ describe("CodeMode schema flexibility", () => {
properties: { id: { type: "string" }, count: { type: "number" } }, properties: { id: { type: "string" }, count: { type: "number" } },
required: ["id"], required: ["id"],
}, },
run: (input) => Effect.sync(() => { run: (input) =>
Effect.sync(() => {
observed.push(input) observed.push(input)
return { echoed: input } return { echoed: input }
}), }),
}) })
const runtime = CodeMode.make({ tools: { adapter: { call } } }) const runtime = CodeMode.make({ tools: { adapter: { call } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "adapter.call", path: "adapter.call",
description: "Call an adapter-described tool", description: "Call an adapter-described tool",
signature: "tools.adapter.call(input: { id: string; count?: number }): Promise<unknown>", signature: "tools.adapter.call(input: { id: string; count?: number }): Promise<unknown>",
}]) },
])
// JSON Schema is render-only: mistyped input passes through unvalidated. // JSON Schema is render-only: mistyped input passes through unvalidated.
const result = await Effect.runPromise(runtime.execute(`return await tools.adapter.call({ id: 42 })`)) const result = await Effect.runPromise(runtime.execute(`return await tools.adapter.call({ id: 42 })`))
@@ -421,17 +451,25 @@ describe("CodeMode schema flexibility", () => {
input: { type: "object", properties: { login: { type: "string" } }, required: ["login"] }, input: { type: "object", properties: { login: { type: "string" } }, required: ["login"] },
output: { output: {
$ref: "#/$defs/User", $ref: "#/$defs/User",
$defs: { User: { type: "object", properties: { login: { type: "string" }, id: { type: "number" } }, required: ["login", "id"] } }, $defs: {
User: {
type: "object",
properties: { login: { type: "string" }, id: { type: "number" } },
required: ["login", "id"],
},
},
}, },
run: () => Effect.succeed({ login: "kit", id: 7 }), run: () => Effect.succeed({ login: "kit", id: 7 }),
}) })
const runtime = CodeMode.make({ tools: { users: { lookup } } }) const runtime = CodeMode.make({ tools: { users: { lookup } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "users.lookup", path: "users.lookup",
description: "Look up a user", description: "Look up a user",
signature: "tools.users.lookup(input: { login: string }): Promise<{ login: string; id: number }>", signature: "tools.users.lookup(input: { login: string }): Promise<{ login: string; id: number }>",
}]) },
])
const result = await Effect.runPromise(runtime.execute(`return await tools.users.lookup({ login: "kit" })`)) const result = await Effect.runPromise(runtime.execute(`return await tools.users.lookup({ login: "kit" })`))
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
@@ -478,16 +516,20 @@ describe("CodeMode public contract", () => {
expect(agentTool.input).toBe(ExecuteInputSchema) expect(agentTool.input).toBe(ExecuteInputSchema)
expect(agentTool.output).toBe(ExecuteResultSchema) expect(agentTool.output).toBe(ExecuteResultSchema)
expect(agentTool.description).toBe(runtime.instructions()) expect(agentTool.description).toBe(runtime.instructions())
expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(projected)))).toStrictEqual(projected) expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(projected)))).toStrictEqual(
projected,
)
}) })
test("inlines a COMPLETE small catalog and keeps search registered but unadvertised", async () => { test("inlines a COMPLETE small catalog and keeps search registered but unadvertised", async () => {
const runtime = CodeMode.make({ tools }) const runtime = CodeMode.make({ tools })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "orders.lookup", path: "orders.lookup",
description: "Look up an order by ID", description: "Look up an order by ID",
signature: "tools.orders.lookup(input: { id: string }): Promise<{ id: string; status: string }>", signature: "tools.orders.lookup(input: { id: string }): Promise<{ id: string; status: string }>",
}]) },
])
expect(runtime.instructions()).toContain("Available tools (COMPLETE list") expect(runtime.instructions()).toContain("Available tools (COMPLETE list")
expect(runtime.instructions()).toContain("- orders (1 tool)") expect(runtime.instructions()).toContain("- orders (1 tool)")
expect(runtime.instructions()).toContain( expect(runtime.instructions()).toContain(
@@ -502,11 +544,13 @@ describe("CodeMode public contract", () => {
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (result.ok) { if (result.ok) {
expect(result.value).toStrictEqual({ expect(result.value).toStrictEqual({
items: [{ items: [
{
path: "tools.orders.lookup", path: "tools.orders.lookup",
description: "Look up an order by ID", description: "Look up an order by ID",
signature: "tools.orders.lookup(input: {\n id: string\n}): Promise<{\n id: string\n status: string\n}>", signature: "tools.orders.lookup(input: {\n id: string\n}): Promise<{\n id: string\n status: string\n}>",
}], },
],
total: 1, total: 1,
}) })
} }
@@ -521,31 +565,43 @@ describe("CodeMode public contract", () => {
}) })
const runtime = CodeMode.make({ tools: { context7: { "resolve-library-id": resolveLibrary } } }) const runtime = CodeMode.make({ tools: { context7: { "resolve-library-id": resolveLibrary } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "context7.resolve-library-id", path: "context7.resolve-library-id",
description: "Resolve a library ID", description: "Resolve a library ID",
signature: 'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>', signature: 'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>',
}]) },
expect(runtime.instructions()).toContain('tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>') ])
expect(runtime.instructions()).toContain(
'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>',
)
const search = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: "resolve library id" })`)) const search = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: "resolve library id" })`),
)
expect(search.ok).toBe(true) expect(search.ok).toBe(true)
if (search.ok) { if (search.ok) {
expect(search.value).toStrictEqual({ expect(search.value).toStrictEqual({
items: [{ items: [
{
path: 'tools.context7["resolve-library-id"]', path: 'tools.context7["resolve-library-id"]',
description: "Resolve a library ID", description: "Resolve a library ID",
signature: 'tools.context7["resolve-library-id"](input: {\n libraryName: string\n}): Promise<string>', signature: 'tools.context7["resolve-library-id"](input: {\n libraryName: string\n}): Promise<string>',
}], },
],
total: 1, total: 1,
}) })
} }
const call = await Effect.runPromise(runtime.execute(`return await tools.context7["resolve-library-id"]({ libraryName: "TypeScript" })`)) const call = await Effect.runPromise(
runtime.execute(`return await tools.context7["resolve-library-id"]({ libraryName: "TypeScript" })`),
)
expect(call.ok).toBe(true) expect(call.ok).toBe(true)
if (call.ok) expect(call.value).toBe("/resolved/TypeScript") if (call.ok) expect(call.value).toBe("/resolved/TypeScript")
const exact = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: 'tools.context7["resolve-library-id"]' })`)) const exact = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: 'tools.context7["resolve-library-id"]' })`),
)
expect(exact.ok).toBe(true) expect(exact.ok).toBe(true)
if (exact.ok) expect((exact.value as { total: number }).total).toBe(1) if (exact.ok) expect((exact.value as { total: number }).total).toBe(1)
}) })
@@ -561,7 +617,9 @@ describe("CodeMode public contract", () => {
expect(instructions.indexOf("## Rules")).toBeLessThan(instructions.indexOf("## Syntax")) expect(instructions.indexOf("## Rules")).toBeLessThan(instructions.indexOf("## Syntax"))
expect(instructions.indexOf("## Syntax")).toBeLessThan(instructions.indexOf("\n## Available tools (COMPLETE list")) expect(instructions.indexOf("## Syntax")).toBeLessThan(instructions.indexOf("\n## Available tools (COMPLETE list"))
// The workflow carries the result-shape guidance; Rules only add content beyond it. // The workflow carries the result-shape guidance; Rules only add content beyond it.
expect(instructions).toContain('`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string') expect(instructions).toContain(
'`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string',
)
expect(instructions).toContain("Return only the fields you need") expect(instructions).toContain("Return only the fields you need")
expect(instructions).toContain("raw payloads get truncated and waste context") expect(instructions).toContain("raw payloads get truncated and waste context")
expect(instructions).toContain("`const res = await tools.<namespace>.<tool>(input)`") expect(instructions).toContain("`const res = await tools.<namespace>.<tool>(input)`")
@@ -584,8 +642,12 @@ describe("CodeMode public contract", () => {
expect(partial).toContain( expect(partial).toContain(
'1. Find a tool (skip when it is already listed below): `const { items } = await tools.$codemode.search({ query: "<intent + key nouns>" })` — short phrases like "list issues" work best.', '1. Find a tool (skip when it is already listed below): `const { items } = await tools.$codemode.search({ query: "<intent + key nouns>" })` — short phrases like "list issues" work best.',
) )
expect(partial).toContain("Only tools listed here or returned by `tools.$codemode.search` are available inside `tools`") expect(partial).toContain(
expect(partial).toContain('- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.') "Only tools listed here or returned by `tools.$codemode.search` are available inside `tools`",
)
expect(partial).toContain(
'- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.',
)
expect(partial).not.toContain("total_count") expect(partial).not.toContain("total_count")
expect(partial).not.toContain("tools.orders.lookup({") expect(partial).not.toContain("tools.orders.lookup({")
}) })
@@ -604,7 +666,9 @@ describe("CodeMode public contract", () => {
expect(instructions).not.toContain("instanceof Error") expect(instructions).not.toContain("instanceof Error")
expect(instructions).not.toContain("splice") expect(instructions).not.toContain("splice")
// The data-boundary note survives. // The data-boundary note survives.
expect(instructions).toContain("Dates serialize to ISO strings at data boundaries; Map/Set/RegExp serialize to `{}`.") expect(instructions).toContain(
"Dates serialize to ISO strings at data boundaries; Map/Set/RegExp serialize to `{}`.",
)
}) })
test("zero tools keep minimal sections and the no-tools notice", () => { test("zero tools keep minimal sections and the no-tools notice", () => {
@@ -635,18 +699,22 @@ describe("CodeMode public contract", () => {
tools: { thread: { uploadFile: upload, generateImage: generate }, orders: { lookup } }, tools: { thread: { uploadFile: upload, generateImage: generate }, orders: { lookup } },
discovery: { maxInlineCatalogTokens: 0 }, discovery: { maxInlineCatalogTokens: 0 },
}) })
expect(runtime.instructions()).toContain("Available tools (PARTIAL — 0 of 3 shown; find the rest with tools.$codemode.search)") expect(runtime.instructions()).toContain(
"Available tools (PARTIAL — 0 of 3 shown; find the rest with tools.$codemode.search)",
)
expect(runtime.instructions()).toContain("- thread (2 tools, none shown)") expect(runtime.instructions()).toContain("- thread (2 tools, none shown)")
expect(runtime.instructions()).toContain("- orders (1 tool, none shown)") expect(runtime.instructions()).toContain("- orders (1 tool, none shown)")
expect(runtime.instructions()).toMatch(/\$codemode\.search/) expect(runtime.instructions()).toMatch(/\$codemode\.search/)
expect(runtime.instructions()).not.toMatch(/tools\.thread\.uploadFile\(input/) expect(runtime.instructions()).not.toMatch(/tools\.thread\.uploadFile\(input/)
const result = await Effect.runPromise(runtime.execute(` const result = await Effect.runPromise(
runtime.execute(`
return await tools.$codemode.search({ return await tools.$codemode.search({
query: "send message attachment upload file to current Discord thread", query: "send message attachment upload file to current Discord thread",
limit: 2 limit: 2
}) })
`)) `),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
expect(result.value).toStrictEqual({ expect(result.value).toStrictEqual({
@@ -666,19 +734,27 @@ describe("CodeMode public contract", () => {
}) })
expect(result.toolCalls).toStrictEqual([{ name: "$codemode.search" }]) expect(result.toolCalls).toStrictEqual([{ name: "$codemode.search" }])
const variants = await Effect.runPromise(runtime.execute(` const variants = await Effect.runPromise(
runtime.execute(`
return await Promise.all([ return await Promise.all([
tools.$codemode.search({ query: "file" }), tools.$codemode.search({ query: "file" }),
tools.$codemode.search({ query: "image" }) tools.$codemode.search({ query: "image" })
]) ])
`)) `),
)
expect(variants.ok).toBe(true) expect(variants.ok).toBe(true)
if (variants.ok) { if (variants.ok) {
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[0]?.items[0]?.path).toBe("tools.thread.uploadFile") expect((variants.value as Array<{ items: Array<{ path: string }> }>)[0]?.items[0]?.path).toBe(
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[1]?.items[0]?.path).toBe("tools.thread.generateImage") "tools.thread.uploadFile",
)
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[1]?.items[0]?.path).toBe(
"tools.thread.generateImage",
)
} }
const removed = await Effect.runPromise(runtime.execute(`return await tools.$codemode.describe({ path: "thread.uploadFile" })`)) const removed = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.describe({ path: "thread.uploadFile" })`),
)
expect(removed.ok).toBe(false) expect(removed.ok).toBe(false)
if (!removed.ok) expect(removed.error.kind).toBe("UnknownTool") if (!removed.ok) expect(removed.error.kind).toBe("UnknownTool")
}) })
@@ -706,15 +782,19 @@ describe("CodeMode public contract", () => {
} }
for (const query of ["many.tool13", "tools.many.tool13"]) { for (const query of ["many.tool13", "tools.many.tool13"]) {
const exact = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`)) const exact = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`),
)
expect(exact.ok).toBe(true) expect(exact.ok).toBe(true)
if (exact.ok) { if (exact.ok) {
expect(exact.value).toStrictEqual({ expect(exact.value).toStrictEqual({
items: [{ items: [
{
path: "tools.many.tool13", path: "tools.many.tool13",
description: "Numbered tool 13", description: "Numbered tool 13",
signature: "tools.many.tool13(input: {\n id: string\n}): Promise<string>", signature: "tools.many.tool13(input: {\n id: string\n}): Promise<string>",
}], },
],
total: 1, total: 1,
}) })
} }
@@ -737,20 +817,23 @@ describe("CodeMode public contract", () => {
}) })
// Empty query + namespace browses just that namespace, alphabetical by path. // Empty query + namespace browses just that namespace, alphabetical by path.
const browse = await Effect.runPromise(runtime.execute( const browse = await Effect.runPromise(
`return await tools.$codemode.search({ query: "", namespace: "github" })`, runtime.execute(`return await tools.$codemode.search({ query: "", namespace: "github" })`),
)) )
expect(browse.ok).toBe(true) expect(browse.ok).toBe(true)
if (browse.ok) { if (browse.ok) {
const value = browse.value as { items: Array<{ path: string }>; total: number } const value = browse.value as { items: Array<{ path: string }>; total: number }
expect(value.total).toBe(2) expect(value.total).toBe(2)
expect(value.items.map((item) => item.path)).toStrictEqual(["tools.github.create_issue", "tools.github.list_issues"]) expect(value.items.map((item) => item.path)).toStrictEqual([
"tools.github.create_issue",
"tools.github.list_issues",
])
} }
// A query + namespace ranks within that namespace only. // A query + namespace ranks within that namespace only.
const scoped = await Effect.runPromise(runtime.execute( const scoped = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: "linear" })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: "linear" })`),
)) )
expect(scoped.ok).toBe(true) expect(scoped.ok).toBe(true)
if (scoped.ok) { if (scoped.ok) {
const value = scoped.value as { items: Array<{ path: string }>; total: number } const value = scoped.value as { items: Array<{ path: string }>; total: number }
@@ -758,9 +841,9 @@ describe("CodeMode public contract", () => {
expect(value.items[0]?.path).toBe("tools.linear.list_issues") expect(value.items[0]?.path).toBe("tools.linear.list_issues")
} }
const invalid = await Effect.runPromise(runtime.execute( const invalid = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: 7 })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: 7 })`),
)) )
expect(invalid.ok).toBe(false) expect(invalid.ok).toBe(false)
if (!invalid.ok) expect(invalid.error.kind).toBe("InvalidToolInput") if (!invalid.ok) expect(invalid.error.kind).toBe("InvalidToolInput")
}) })
@@ -785,9 +868,9 @@ describe("CodeMode public contract", () => {
// "attachment" appears in neither path nor description — only in the input schema's // "attachment" appears in neither path nor description — only in the input schema's
// property names, which the searchable text includes. // property names, which the searchable text includes.
const byParameter = await Effect.runPromise(runtime.execute( const byParameter = await Effect.runPromise(
`return await tools.$codemode.search({ query: "attachment" })`, runtime.execute(`return await tools.$codemode.search({ query: "attachment" })`),
)) )
expect(byParameter.ok).toBe(true) expect(byParameter.ok).toBe(true)
if (byParameter.ok) { if (byParameter.ok) {
const value = byParameter.value as { items: Array<{ path: string }>; total: number } const value = byParameter.value as { items: Array<{ path: string }>; total: number }
@@ -796,9 +879,9 @@ describe("CodeMode public contract", () => {
} }
// Substring matching: a partial word ("docum") still hits the description. // Substring matching: a partial word ("docum") still hits the description.
const bySubstring = await Effect.runPromise(runtime.execute( const bySubstring = await Effect.runPromise(
`return await tools.$codemode.search({ query: "docum" })`, runtime.execute(`return await tools.$codemode.search({ query: "docum" })`),
)) )
expect(bySubstring.ok).toBe(true) expect(bySubstring.ok).toBe(true)
if (bySubstring.ok) { if (bySubstring.ok) {
const value = bySubstring.value as { items: Array<{ path: string }>; total: number } const value = bySubstring.value as { items: Array<{ path: string }>; total: number }
@@ -825,9 +908,9 @@ describe("CodeMode public contract", () => {
}) })
// "issues" still finds the singular-only tool (term OR singular(term) per field)... // "issues" still finds the singular-only tool (term OR singular(term) per field)...
const plural = await Effect.runPromise(runtime.execute( const plural = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: "tracker" })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: "tracker" })`),
)) )
expect(plural.ok).toBe(true) expect(plural.ok).toBe(true)
if (plural.ok) { if (plural.ok) {
const value = plural.value as { items: Array<{ path: string }>; total: number } const value = plural.value as { items: Array<{ path: string }>; total: number }
@@ -836,14 +919,15 @@ describe("CodeMode public contract", () => {
} }
// ...while a true "issues" path match still outranks the singular-only description match. // ...while a true "issues" path match still outranks the singular-only description match.
const ranked = await Effect.runPromise(runtime.execute( const ranked = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: "issues" })`))
`return await tools.$codemode.search({ query: "issues" })`,
))
expect(ranked.ok).toBe(true) expect(ranked.ok).toBe(true)
if (ranked.ok) { if (ranked.ok) {
const value = ranked.value as { items: Array<{ path: string }>; total: number } const value = ranked.value as { items: Array<{ path: string }>; total: number }
expect(value.total).toBe(2) expect(value.total).toBe(2)
expect(value.items.map((item) => item.path)).toStrictEqual(["tools.github.list_issues", "tools.tracker.fetch_all"]) expect(value.items.map((item) => item.path)).toStrictEqual([
"tools.github.list_issues",
"tools.tracker.fetch_all",
])
} }
}) })
@@ -882,8 +966,12 @@ describe("CodeMode public contract", () => {
run: () => Effect.succeed("ok"), run: () => Effect.succeed("ok"),
}) })
const expensive = Tool.make({ const expensive = Tool.make({
description: "An expensive tool whose description alone consumes far more than the remaining inline catalog byte budget for this runtime", description:
input: Schema.Struct({ someRatherLongParameterName: Schema.String, anotherEvenLongerParameterName: Schema.Number }), "An expensive tool whose description alone consumes far more than the remaining inline catalog byte budget for this runtime",
input: Schema.Struct({
someRatherLongParameterName: Schema.String,
anotherEvenLongerParameterName: Schema.Number,
}),
output: Schema.String, output: Schema.String,
run: () => Effect.succeed("ok"), run: () => Effect.succeed("ok"),
}) })
@@ -896,7 +984,9 @@ describe("CodeMode public contract", () => {
}) })
const instructions = runtime.instructions() const instructions = runtime.instructions()
expect(instructions).toContain("Available tools (PARTIAL — 2 of 3 shown; find the rest with tools.$codemode.search)") expect(instructions).toContain(
"Available tools (PARTIAL — 2 of 3 shown; find the rest with tools.$codemode.search)",
)
expect(instructions).toContain("- alpha (2 tools, 1 shown)") expect(instructions).toContain("- alpha (2 tools, 1 shown)")
expect(instructions).toContain(" - tools.alpha.cheap(input: { q: string }): Promise<string> // Cheap") expect(instructions).toContain(" - tools.alpha.cheap(input: { q: string }): Promise<string> // Cheap")
expect(instructions).not.toContain("tools.alpha.expensive(") expect(instructions).not.toContain("tools.alpha.expensive(")
@@ -912,7 +1002,8 @@ describe("CodeMode public contract", () => {
description: "Double a number", description: "Double a number",
input: Schema.Struct({ value: Schema.NumberFromString }), input: Schema.Struct({ value: Schema.NumberFromString }),
output: Schema.NumberFromString, output: Schema.NumberFromString,
run: ({ value }) => Effect.sync(() => { run: ({ value }) =>
Effect.sync(() => {
observed.push(value) observed.push(value)
return String(value * 2) return String(value * 2)
}), }),
@@ -934,9 +1025,11 @@ describe("CodeMode public contract", () => {
}) })
test("returns JSON-safe data and normalizes undefined to null", async () => { test("returns JSON-safe data and normalizes undefined to null", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: `return { top: undefined, nested: [1, undefined] }`, code: `return { top: undefined, nested: [1, undefined] }`,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: { top: null, nested: [1, null] }, value: { top: null, nested: [1, null] },
@@ -947,18 +1040,20 @@ describe("CodeMode public contract", () => {
test("rejects invalid configuration and discovery limits", async () => { test("rejects invalid configuration and discovery limits", async () => {
expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: 0 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: 0 } })).toThrow(RangeError)
expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: Number.POSITIVE_INFINITY } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: Number.POSITIVE_INFINITY } })).toThrow(
RangeError,
)
expect(() => CodeMode.execute({ code: "return 1", limits: { maxToolCalls: -1 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { maxToolCalls: -1 } })).toThrow(RangeError)
expect(() => CodeMode.execute({ code: "return 1", limits: { maxOutputBytes: -1 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { maxOutputBytes: -1 } })).toThrow(RangeError)
expect(() => CodeMode.make({ tools, discovery: { maxInlineCatalogTokens: -1 } })).toThrow(RangeError) expect(() => CodeMode.make({ tools, discovery: { maxInlineCatalogTokens: -1 } })).toThrow(RangeError)
const result = await Effect.runPromise(CodeMode.make({ const result = await Effect.runPromise(
CodeMode.make({
tools, tools,
discovery: { maxInlineCatalogTokens: 0 }, discovery: { maxInlineCatalogTokens: 0 },
}).execute( }).execute(`return await tools.$codemode.search({ query: "order", limit: 0.5 })`),
`return await tools.$codemode.search({ query: "order", limit: 0.5 })`, )
))
expect(result.ok).toBe(false) expect(result.ok).toBe(false)
if (result.ok) return if (result.ok) return
expect(result.error.kind).toBe("InvalidToolInput") expect(result.error.kind).toBe("InvalidToolInput")
@@ -979,14 +1074,16 @@ describe("CodeMode public contract", () => {
output: Schema.Number, output: Schema.Number,
run: () => Effect.succeed(1), run: () => Effect.succeed(1),
}) })
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
tools: { host: { count: counter } }, tools: { host: { count: counter } },
code: ` code: `
let total = 0 let total = 0
for (let i = 0; i < 150; i += 1) total += await tools.host.count({}) for (let i = 0; i < 150; i += 1) total += await tools.host.count({})
return total return total
`, `,
})) }),
)
expect(result).toMatchObject({ ok: true, value: 150 }) expect(result).toMatchObject({ ok: true, value: 150 })
if (result.ok) expect(result.toolCalls.length).toBe(150) if (result.ok) expect(result.toolCalls.length).toBe(150)
}) })
@@ -1008,8 +1105,6 @@ describe("CodeMode public contract", () => {
}) })
test("reserves the discovery namespace", () => { test("reserves the discovery namespace", () => {
expect(() => CodeMode.make({ tools: { $codemode: { lookup } } })).toThrow( expect(() => CodeMode.make({ tools: { $codemode: { lookup } } })).toThrow(/reserved for CodeMode discovery tools/)
/reserved for CodeMode discovery tools/,
)
}) })
}) })
+24 -12
View File
@@ -36,10 +36,12 @@ const error = async (code: string) => {
describe("Object.keys over tool references", () => { describe("Object.keys over tool references", () => {
test("enumerates top-level namespaces (the transcript program)", async () => { test("enumerates top-level namespaces (the transcript program)", async () => {
expect(await value(` expect(
await value(`
const namespaces = Object.keys(tools) const namespaces = Object.keys(tools)
return { namespaces, count: namespaces.length } return { namespaces, count: namespaces.length }
`)).toEqual({ namespaces: ["github", "memory", "playwright"], count: 3 }) `),
).toEqual({ namespaces: ["github", "memory", "playwright"], count: 3 })
}) })
test("enumerates tool names at a nested namespace", async () => { test("enumerates tool names at a nested namespace", async () => {
@@ -92,7 +94,8 @@ describe("Object.keys over arrays", () => {
describe("for...in", () => { describe("for...in", () => {
test("iterates own enumerable keys of a plain object with break/continue", async () => { test("iterates own enumerable keys of a plain object with break/continue", async () => {
expect(await value(` expect(
await value(`
const seen = [] const seen = []
for (const key in { a: 1, b: 2, c: 3, d: 4 }) { for (const key in { a: 1, b: 2, c: 3, d: 4 }) {
if (key === "b") continue if (key === "b") continue
@@ -100,41 +103,50 @@ describe("for...in", () => {
seen.push(key) seen.push(key)
} }
return seen return seen
`)).toEqual(["a", "c"]) `),
).toEqual(["a", "c"])
}) })
test("iterates index strings over arrays", async () => { test("iterates index strings over arrays", async () => {
expect(await value(` expect(
await value(`
const indexes = [] const indexes = []
for (const i in ["x", "y", "z"]) { for (const i in ["x", "y", "z"]) {
if (i === "2") break if (i === "2") break
indexes.push(i) indexes.push(i)
} }
return indexes return indexes
`)).toEqual(["0", "1"]) `),
).toEqual(["0", "1"])
}) })
test("supports let declarations and bare identifiers", async () => { test("supports let declarations and bare identifiers", async () => {
expect(await value(` expect(
await value(`
let last = "" let last = ""
for (let key in { a: 1, b: 2 }) last = key for (let key in { a: 1, b: 2 }) last = key
return last return last
`)).toBe("b") `),
expect(await value(` ).toBe("b")
expect(
await value(`
let key = "before" let key = "before"
for (key in { only: 1 }) {} for (key in { only: 1 }) {}
return key return key
`)).toBe("only") `),
).toBe("only")
}) })
test("enumerates namespaces and tools from the host tool tree", async () => { test("enumerates namespaces and tools from the host tool tree", async () => {
expect(await value(` expect(
await value(`
const names = [] const names = []
for (const ns in tools) { for (const ns in tools) {
for (const name in tools[ns]) names.push(ns + "." + name) for (const name in tools[ns]) names.push(ns + "." + name)
} }
return names return names
`)).toEqual(["github.list_issues", "github.get_issue", "memory.search", "playwright.navigate"]) `),
).toEqual(["github.list_issues", "github.get_issue", "memory.search", "playwright.navigate"])
}) })
test("unsupported values fail with a hint at for...of and Object.keys", async () => { test("unsupported values fail with a hint at for...of and Object.keys", async () => {
+95 -50
View File
@@ -143,21 +143,40 @@ describe("H1: NaN/Infinity flow as intermediates and normalize to null at the bo
describe("Error values and instanceof", () => { describe("Error values and instanceof", () => {
test("new Error carries name/message and is instanceof Error", async () => { test("new Error carries name/message and is instanceof Error", async () => {
expect(await value(`const e = new Error("boom"); return [e instanceof Error, e.name, e.message]`)).toEqual([true, "Error", "boom"]) expect(await value(`const e = new Error("boom"); return [e instanceof Error, e.name, e.message]`)).toEqual([
true,
"Error",
"boom",
])
}) })
test("Error without new behaves like new Error", async () => { test("Error without new behaves like new Error", async () => {
expect(await value(`const e = Error("plain"); return [e instanceof Error, e.name, e.message]`)).toEqual([true, "Error", "plain"]) expect(await value(`const e = Error("plain"); return [e instanceof Error, e.name, e.message]`)).toEqual([
expect(await value(`const e = new Error(); return [e.name, e.message, e instanceof Error]`)).toEqual(["Error", "", true]) true,
"Error",
"plain",
])
expect(await value(`const e = new Error(); return [e.name, e.message, e instanceof Error]`)).toEqual([
"Error",
"",
true,
])
}) })
test("specific error types are instanceof themselves and Error, not each other", async () => { test("specific error types are instanceof themselves and Error, not each other", async () => {
expect(await value(`const e = new TypeError("t"); return [e instanceof TypeError, e instanceof Error, e instanceof RangeError]`)).toEqual([true, true, false]) expect(
await value(
`const e = new TypeError("t"); return [e instanceof TypeError, e instanceof Error, e instanceof RangeError]`,
),
).toEqual([true, true, false])
expect(await value(`return new Error("e") instanceof TypeError`)).toBe(false) expect(await value(`return new Error("e") instanceof TypeError`)).toBe(false)
}) })
test("thrown errors keep instanceof through try/catch", async () => { test("thrown errors keep instanceof through try/catch", async () => {
expect(await value(`try { throw new Error("x") } catch (e) { return [e instanceof Error, e.message] }`)).toEqual([true, "x"]) expect(await value(`try { throw new Error("x") } catch (e) { return [e instanceof Error, e.message] }`)).toEqual([
true,
"x",
])
}) })
test("interpreter runtime failures are caught as Error values", async () => { test("interpreter runtime failures are caught as Error values", async () => {
@@ -168,33 +187,48 @@ describe("Error values and instanceof", () => {
test("caught failures carry the constructor name the real-JS failure would have", async () => { test("caught failures carry the constructor name the real-JS failure would have", async () => {
// JSON.parse throws SyntaxError: name and specific-instanceof both carry through, and the // JSON.parse throws SyntaxError: name and specific-instanceof both carry through, and the
// message keeps the engine's position detail. // message keeps the engine's position detail.
expect(await value(` expect(
await value(`
try { JSON.parse("{oops") } catch (e) { try { JSON.parse("{oops") } catch (e) {
return [e.name, e instanceof SyntaxError, e instanceof Error, e instanceof TypeError, e.message.includes("JSON")] return [e.name, e instanceof SyntaxError, e instanceof Error, e instanceof TypeError, e.message.includes("JSON")]
} }
`)).toEqual(["SyntaxError", true, true, false, true]) `),
expect(await value(`try { undeclared() } catch (e) { return [e.name, e instanceof ReferenceError] }`)) ).toEqual(["SyntaxError", true, true, false, true])
.toEqual(["ReferenceError", true]) expect(await value(`try { undeclared() } catch (e) { return [e.name, e instanceof ReferenceError] }`)).toEqual([
expect(await value(`try { const c = 1; c = 2 } catch (e) { return [e.name, e instanceof TypeError] }`)) "ReferenceError",
.toEqual(["TypeError", true]) true,
expect(await value(`try { "a".normalize("NOPE") } catch (e) { return [e.name, e instanceof RangeError] }`)) ])
.toEqual(["RangeError", true]) expect(await value(`try { const c = 1; c = 2 } catch (e) { return [e.name, e instanceof TypeError] }`)).toEqual([
expect(await value(`try { "a".match("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)) "TypeError",
.toEqual(["SyntaxError", true]) true,
expect(await value(`try { new RegExp("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)) ])
.toEqual(["SyntaxError", true]) expect(await value(`try { "a".normalize("NOPE") } catch (e) { return [e.name, e instanceof RangeError] }`)).toEqual(
["RangeError", true],
)
expect(await value(`try { "a".match("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)).toEqual([
"SyntaxError",
true,
])
expect(await value(`try { new RegExp("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)).toEqual([
"SyntaxError",
true,
])
}) })
test("diagnostics without a specific real-JS analogue are named plain Error", async () => { test("diagnostics without a specific real-JS analogue are named plain Error", async () => {
expect(await value(`try { JSON.parse(5) } catch (e) { return [e.name, e instanceof Error] }`)) expect(await value(`try { JSON.parse(5) } catch (e) { return [e.name, e instanceof Error] }`)).toEqual([
.toEqual(["Error", true]) "Error",
true,
])
}) })
test("Promise.allSettled rejection reasons are Error values", async () => { test("Promise.allSettled rejection reasons are Error values", async () => {
expect(await value(` expect(
await value(`
const settled = await Promise.allSettled([Promise.reject(new Error("b"))]) const settled = await Promise.allSettled([Promise.reject(new Error("b"))])
return [settled[0].reason instanceof Error, settled[0].reason.message] return [settled[0].reason instanceof Error, settled[0].reason.message]
`)).toEqual([true, "b"]) `),
).toEqual([true, "b"])
}) })
test("non-error thrown values are not instanceof Error", async () => { test("non-error thrown values are not instanceof Error", async () => {
@@ -203,7 +237,11 @@ describe("Error values and instanceof", () => {
}) })
test("plain data is never instanceof Error", async () => { test("plain data is never instanceof Error", async () => {
expect(await value(`return [({}) instanceof Error, "s" instanceof Error, null instanceof Error]`)).toEqual([false, false, false]) expect(await value(`return [({}) instanceof Error, "s" instanceof Error, null instanceof Error]`)).toEqual([
false,
false,
false,
])
}) })
test("error values still serialize as plain { name, message } data", async () => { test("error values still serialize as plain { name, message } data", async () => {
@@ -227,17 +265,29 @@ describe("Error values and instanceof", () => {
describe("array methods: splice, fill, copyWithin, keys/values/entries", () => { describe("array methods: splice, fill, copyWithin, keys/values/entries", () => {
test("splice removes in place and returns the removed elements", async () => { test("splice removes in place and returns the removed elements", async () => {
expect(await value(`const a = [1,2,3,4]; const removed = a.splice(1, 2); return { removed, a }`)).toEqual({ removed: [2, 3], a: [1, 4] }) expect(await value(`const a = [1,2,3,4]; const removed = a.splice(1, 2); return { removed, a }`)).toEqual({
removed: [2, 3],
a: [1, 4],
})
}) })
test("splice inserts new elements at the cut", async () => { test("splice inserts new elements at the cut", async () => {
expect(await value(`const a = ["a","d"]; a.splice(1, 0, "b", "c"); return a`)).toEqual(["a", "b", "c", "d"]) expect(await value(`const a = ["a","d"]; a.splice(1, 0, "b", "c"); return a`)).toEqual(["a", "b", "c", "d"])
expect(await value(`const a = [1,2,3]; const removed = a.splice(1, 1, "x"); return { removed, a }`)).toEqual({ removed: [2], a: [1, "x", 3] }) expect(await value(`const a = [1,2,3]; const removed = a.splice(1, 1, "x"); return { removed, a }`)).toEqual({
removed: [2],
a: [1, "x", 3],
})
}) })
test("splice with one argument removes to the end; negative start counts back", async () => { test("splice with one argument removes to the end; negative start counts back", async () => {
expect(await value(`const a = [1,2,3]; const removed = a.splice(1); return { removed, a }`)).toEqual({ removed: [2, 3], a: [1] }) expect(await value(`const a = [1,2,3]; const removed = a.splice(1); return { removed, a }`)).toEqual({
expect(await value(`const a = [1,2,3]; const removed = a.splice(-1); return { removed, a }`)).toEqual({ removed: [3], a: [1, 2] }) removed: [2, 3],
a: [1],
})
expect(await value(`const a = [1,2,3]; const removed = a.splice(-1); return { removed, a }`)).toEqual({
removed: [3],
a: [1, 2],
})
}) })
test("splice rejects inserting a container into itself", async () => { test("splice rejects inserting a container into itself", async () => {
@@ -258,11 +308,13 @@ describe("array methods: splice, fill, copyWithin, keys/values/entries", () => {
test("keys/values/entries return arrays usable with for...of and spread", async () => { test("keys/values/entries return arrays usable with for...of and spread", async () => {
expect(await value(`return [...["x","y","z"].keys()]`)).toEqual([0, 1, 2]) expect(await value(`return [...["x","y","z"].keys()]`)).toEqual([0, 1, 2])
expect(await value(`return ["x","y"].values()`)).toEqual(["x", "y"]) expect(await value(`return ["x","y"].values()`)).toEqual(["x", "y"])
expect(await value(` expect(
await value(`
const out = [] const out = []
for (const [index, item] of ["a","b"].entries()) out.push(index + ":" + item) for (const [index, item] of ["a","b"].entries()) out.push(index + ":" + item)
return out return out
`)).toEqual(["0:a", "1:b"]) `),
).toEqual(["0:a", "1:b"])
expect(await value(`return [...[7].entries()]`)).toEqual([[0, 7]]) expect(await value(`return [...[7].entries()]`)).toEqual([[0, 7]])
}) })
}) })
@@ -300,40 +352,33 @@ describe("compound assignment matches its binary operator", () => {
} }
test("sandbox Date += concatenates its string form, like d = d + 1", async () => { test("sandbox Date += concatenates its string form, like d = d + 1", async () => {
const result = await pair( const result = await pair(`let d = new Date(1000); d += 1; return d`, `let d = new Date(1000); d = d + 1; return d`)
`let d = new Date(1000); d += 1; return d`,
`let d = new Date(1000); d = d + 1; return d`,
)
expect(result).toBe("1970-01-01T00:00:01.000Z1") expect(result).toBe("1970-01-01T00:00:01.000Z1")
}) })
test("sandbox Date numeric compound ops use its time value", async () => { test("sandbox Date numeric compound ops use its time value", async () => {
expect(await pair( expect(
`let d = new Date(1000); d -= 400; return d`, await pair(`let d = new Date(1000); d -= 400; return d`, `let d = new Date(1000); d = d - 400; return d`),
`let d = new Date(1000); d = d - 400; return d`, ).toBe(600)
)).toBe(600) expect(await pair(`let d = new Date(1000); d /= 4; return d`, `let d = new Date(1000); d = d / 4; return d`)).toBe(
expect(await pair( 250,
`let d = new Date(1000); d /= 4; return d`, )
`let d = new Date(1000); d = d / 4; return d`,
)).toBe(250)
}) })
test("string += object/array matches x = x + obj", async () => { test("string += object/array matches x = x + obj", async () => {
expect(await pair( expect(await pair(`let x = "a"; x += { b: 1 }; return x`, `let x = "a"; x = x + { b: 1 }; return x`)).toBe(
`let x = "a"; x += { b: 1 }; return x`, "a[object Object]",
`let x = "a"; x = x + { b: 1 }; return x`, )
)).toBe("a[object Object]") expect(await pair(`let x = "a"; x += [1, 2]; return x`, `let x = "a"; x = x + [1, 2]; return x`)).toBe("a1,2")
expect(await pair(
`let x = "a"; x += [1, 2]; return x`,
`let x = "a"; x = x + [1, 2]; return x`,
)).toBe("a1,2")
}) })
test("compound assignment through a member target coerces the same way", async () => { test("compound assignment through a member target coerces the same way", async () => {
expect(await pair( expect(
await pair(
`const o = { s: "t" }; o.s += new Date(0); return o.s`, `const o = { s: "t" }; o.s += new Date(0); return o.s`,
`const o = { s: "t" }; o.s = o.s + new Date(0); return o.s`, `const o = { s: "t" }; o.s = o.s + new Date(0); return o.s`,
)).toBe("t1970-01-01T00:00:00.000Z") ),
).toBe("t1970-01-01T00:00:00.000Z")
}) })
test("numeric and string compound operators sweep identically to their expansions", async () => { test("numeric and string compound operators sweep identically to their expansions", async () => {
+48 -21
View File
@@ -31,10 +31,14 @@ const sleepyTool = (trace: Trace) =>
trace.active -= 1 trace.active -= 1
trace.completed += 1 trace.completed += 1
return id return id
}).pipe(Effect.onInterrupt(() => Effect.sync(() => { }).pipe(
Effect.onInterrupt(() =>
Effect.sync(() => {
trace.active -= 1 trace.active -= 1
trace.interrupted += 1 trace.interrupted += 1
}))), }),
),
),
}) })
const failingTool = Tool.make({ const failingTool = Tool.make({
@@ -46,11 +50,13 @@ const failingTool = Tool.make({
const run = (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}): Promise<ExecuteResult> => { const run = (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}): Promise<ExecuteResult> => {
const trace = options.trace ?? makeTrace() const trace = options.trace ?? makeTrace()
return Effect.runPromise(CodeMode.execute({ return Effect.runPromise(
CodeMode.execute({
tools: { host: { sleepy: sleepyTool(trace), fail: failingTool } }, tools: { host: { sleepy: sleepyTool(trace), fail: failingTool } },
code, code,
...(options.limits ? { limits: options.limits } : {}), ...(options.limits ? { limits: options.limits } : {}),
})) }),
)
} }
const value = async (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}) => { const value = async (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}) => {
@@ -121,7 +127,8 @@ describe("first-class promise values", () => {
}) })
test("an awaited failure is catchable exactly like a synchronous throw", async () => { test("an awaited failure is catchable exactly like a synchronous throw", async () => {
expect(await value(` expect(
await value(`
const p = tools.host.fail({}) const p = tools.host.fail({})
try { try {
await p await p
@@ -129,7 +136,8 @@ describe("first-class promise values", () => {
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a fire-and-forget call completes before the execution ends", async () => { test("a fire-and-forget call completes before the execution ends", async () => {
@@ -186,20 +194,24 @@ describe("promises at data boundaries", () => {
describe("Promise.all over arbitrary arrays", () => { describe("Promise.all over arbitrary arrays", () => {
test("mixes promises and plain values, preserving order", async () => { test("mixes promises and plain values, preserving order", async () => {
expect(await value(` expect(
await value(`
return await Promise.all([tools.host.sleepy({ id: 1 }), "plain", tools.host.sleepy({ id: 2 }), 42]) return await Promise.all([tools.host.sleepy({ id: 1 }), "plain", tools.host.sleepy({ id: 2 }), 42])
`)).toEqual([1, "plain", 2, 42]) `),
).toEqual([1, "plain", 2, 42])
}) })
test("accepts arrays built beforehand, passed as identifiers, and spread elements", async () => { test("accepts arrays built beforehand, passed as identifiers, and spread elements", async () => {
expect(await value(` expect(
await value(`
const calls = [] const calls = []
calls.push(tools.host.sleepy({ id: 1 })) calls.push(tools.host.sleepy({ id: 1 }))
calls.push(7) calls.push(7)
const more = [tools.host.sleepy({ id: 2 })] const more = [tools.host.sleepy({ id: 2 })]
const batch = [...calls, ...more, "x"] const batch = [...calls, ...more, "x"]
return await Promise.all(batch) return await Promise.all(batch)
`)).toEqual([1, 7, 2, "x"]) `),
).toEqual([1, 7, 2, "x"])
}) })
test("runs items.map tool calls in parallel", async () => { test("runs items.map tool calls in parallel", async () => {
@@ -238,14 +250,16 @@ describe("Promise.all over arbitrary arrays", () => {
}) })
test("rejects with the first failure, catchable in-program", async () => { test("rejects with the first failure, catchable in-program", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.all([tools.host.sleepy({ id: 1 }), tools.host.fail({})]) await Promise.all([tools.host.sleepy({ id: 1 }), tools.host.fail({})])
return "no" return "no"
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a non-collection argument is a clear error", async () => { test("a non-collection argument is a clear error", async () => {
@@ -264,14 +278,16 @@ describe("Promise.all over arbitrary arrays", () => {
describe("Promise.allSettled", () => { describe("Promise.allSettled", () => {
test("reports fulfilled and rejected outcomes with catch-normalized reasons", async () => { test("reports fulfilled and rejected outcomes with catch-normalized reasons", async () => {
expect(await value(` expect(
await value(`
return await Promise.allSettled([ return await Promise.allSettled([
tools.host.sleepy({ id: 5 }), tools.host.sleepy({ id: 5 }),
tools.host.fail({}), tools.host.fail({}),
"plain", "plain",
Promise.reject(new Error("boom")), Promise.reject(new Error("boom")),
]) ])
`)).toEqual([ `),
).toEqual([
{ status: "fulfilled", value: 5 }, { status: "fulfilled", value: 5 },
{ status: "rejected", reason: { name: "Error", message: "Lookup refused" } }, { status: "rejected", reason: { name: "Error", message: "Lookup refused" } },
{ status: "fulfilled", value: "plain" }, { status: "fulfilled", value: "plain" },
@@ -306,7 +322,8 @@ describe("Promise.race", () => {
}) })
test("awaiting an interrupted loser afterwards is a catchable program failure", async () => { test("awaiting an interrupted loser afterwards is a catchable program failure", async () => {
expect(await value(` expect(
await value(`
const fast = tools.host.sleepy({ id: 1, ms: 10 }) const fast = tools.host.sleepy({ id: 1, ms: 10 })
const slow = tools.host.sleepy({ id: 2, ms: 5000 }) const slow = tools.host.sleepy({ id: 2, ms: 5000 })
const winner = await Promise.race([fast, slow]) const winner = await Promise.race([fast, slow])
@@ -316,23 +333,31 @@ describe("Promise.race", () => {
} catch (e) { } catch (e) {
return { winner, caught: e.message } return { winner, caught: e.message }
} }
`)).toEqual({ winner: 1, caught: "This tool call was interrupted because another value settled a Promise.race first." }) `),
).toEqual({
winner: 1,
caught: "This tool call was interrupted because another value settled a Promise.race first.",
})
}) })
test("a rejection can win the race", async () => { test("a rejection can win the race", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.race([tools.host.fail({}), tools.host.sleepy({ id: 1, ms: 5000 })]) await Promise.race([tools.host.fail({}), tools.host.sleepy({ id: 1, ms: 5000 })])
return "no" return "no"
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a plain value wins over pending promises", async () => { test("a plain value wins over pending promises", async () => {
const trace = makeTrace() const trace = makeTrace()
expect(await value(`return await Promise.race([tools.host.sleepy({ id: 1, ms: 5000 }), "immediate"])`, { trace })).toBe("immediate") expect(
await value(`return await Promise.race([tools.host.sleepy({ id: 1, ms: 5000 }), "immediate"])`, { trace }),
).toBe("immediate")
expect(trace.interrupted).toBe(1) expect(trace.interrupted).toBe(1)
}) })
@@ -350,14 +375,16 @@ describe("Promise.resolve / Promise.reject", () => {
}) })
test("reject produces a promise whose await throws the reason", async () => { test("reject produces a promise whose await throws the reason", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.reject("nope") await Promise.reject("nope")
return "no" return "no"
} catch (e) { } catch (e) {
return e return e
} }
`)).toBe("nope") `),
).toBe("nope")
}) })
}) })
+19 -14
View File
@@ -83,15 +83,9 @@ describe("pretty signature rendering", () => {
true, true,
) )
expect(pretty).toBe( expect(pretty).toBe(
[ ["{", " /** Search filter */", " filter?: {", " /** Issue state */", " state?: string", " }", "}"].join(
"{", "\n",
" /** Search filter */", ),
" filter?: {",
" /** Issue state */",
" state?: string",
" }",
"}",
].join("\n"),
) )
}) })
@@ -119,7 +113,14 @@ describe("pretty signature rendering", () => {
expect(pretty).toContain(" /** @deprecated */\n legacy?: string") expect(pretty).toContain(" /** @deprecated */\n legacy?: string")
expect(pretty).toContain(" /** @format uri */\n homepage?: string") expect(pretty).toContain(" /** @format uri */\n homepage?: string")
expect(pretty).toContain( expect(pretty).toContain(
[" /**", ' * @default ["a","b"]', " * @minItems 2", " * @maxItems 5", " */", " tags?: Array<string>"].join("\n"), [
" /**",
' * @default ["a","b"]',
" * @minItems 2",
" * @maxItems 5",
" */",
" tags?: Array<string>",
].join("\n"),
) )
}) })
@@ -212,7 +213,11 @@ describe("non-identifier property names render as quoted keys", () => {
const tool = Tool.make({ const tool = Tool.make({
description: "Adapter tool with awkward field names", description: "Adapter tool with awkward field names",
input: rawSchema, input: rawSchema,
output: { type: "object", properties: { "content-type": { type: "string" } }, required: ["content-type"] } as const, output: {
type: "object",
properties: { "content-type": { type: "string" } },
required: ["content-type"],
} as const,
run: () => Effect.succeed({ "content-type": "text/plain" }), run: () => Effect.succeed({ "content-type": "text/plain" }),
}) })
expect(inputTypeScript(tool)).toContain('"foo-bar"?: string') expect(inputTypeScript(tool)).toContain('"foo-bar"?: string')
@@ -269,9 +274,9 @@ describe("pretty signatures in search results", () => {
const runtime = CodeMode.make({ tools: { github: { list_issues: listIssues }, orders: { lookup: lookupOrder } } }) const runtime = CodeMode.make({ tools: { github: { list_issues: listIssues }, orders: { lookup: lookupOrder } } })
const search = async (query: string) => { const search = async (query: string) => {
const result = await Effect.runPromise(runtime.execute( const result = await Effect.runPromise(
`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`, runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`),
)) )
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) throw new Error("search failed") if (!result.ok) throw new Error("search failed")
return result.value as { items: Array<{ path: string; signature: string }>; total: number } return result.value as { items: Array<{ path: string; signature: string }>; total: number }
+111 -43
View File
@@ -40,7 +40,11 @@ describe("Date", () => {
}) })
test("UTC getters read calendar components", async () => { test("UTC getters read calendar components", async () => {
expect(await value(`const d = new Date("2024-03-05T06:07:08.009Z"); return [d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate(), d.getUTCHours(), d.getUTCMinutes(), d.getUTCSeconds(), d.getUTCMilliseconds()]`)).toEqual([2024, 2, 5, 6, 7, 8, 9]) expect(
await value(
`const d = new Date("2024-03-05T06:07:08.009Z"); return [d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate(), d.getUTCHours(), d.getUTCMinutes(), d.getUTCSeconds(), d.getUTCMilliseconds()]`,
),
).toEqual([2024, 2, 5, 6, 7, 8, 9])
}) })
test("invalid dates yield NaN times, guardable in-sandbox", async () => { test("invalid dates yield NaN times, guardable in-sandbox", async () => {
@@ -49,7 +53,9 @@ describe("Date", () => {
}) })
test("toISOString on an invalid date is a catchable error", async () => { test("toISOString on an invalid date is a catchable error", async () => {
expect(await value(`try { new Date("garbage").toISOString(); return "no" } catch { return "caught" }`)).toBe("caught") expect(await value(`try { new Date("garbage").toISOString(); return "no" } catch { return "caught" }`)).toBe(
"caught",
)
}) })
test("template interpolation renders the ISO form", async () => { test("template interpolation renders the ISO form", async () => {
@@ -72,14 +78,18 @@ describe("Date", () => {
}) })
test("sorting dates with a numeric comparator", async () => { test("sorting dates with a numeric comparator", async () => {
expect(await value(` expect(
await value(`
const dates = [new Date(3000), new Date(1000), new Date(2000)] const dates = [new Date(3000), new Date(1000), new Date(2000)]
return dates.sort((a, b) => a - b).map((d) => d.getTime()) return dates.sort((a, b) => a - b).map((d) => d.getTime())
`)).toEqual([1000, 2000, 3000]) `),
).toEqual([1000, 2000, 3000])
}) })
test("new Date(year, month, day) accepts component form", async () => { test("new Date(year, month, day) accepts component form", async () => {
expect(await value(`const d = new Date(2024, 0, 2); return [d.getFullYear(), d.getMonth(), d.getDate()]`)).toEqual([2024, 0, 2]) expect(await value(`const d = new Date(2024, 0, 2); return [d.getFullYear(), d.getMonth(), d.getDate()]`)).toEqual([
2024, 0, 2,
])
}) })
test("typeof and unknown properties are forgiving", async () => { test("typeof and unknown properties are forgiving", async () => {
@@ -95,25 +105,31 @@ describe("RegExp", () => {
}) })
test("exec exposes captures and index", async () => { test("exec exposes captures and index", async () => {
expect(await value(`const m = /a(b+)/.exec("xxabbc"); return { full: m[0], group: m[1], index: m.index }`)).toEqual({ expect(await value(`const m = /a(b+)/.exec("xxabbc"); return { full: m[0], group: m[1], index: m.index }`)).toEqual(
{
full: "abb", full: "abb",
group: "bb", group: "bb",
index: 2, index: 2,
}) },
)
expect(await value(`return /a/.exec("zzz")`)).toBeNull() expect(await value(`return /a/.exec("zzz")`)).toBeNull()
}) })
test("named groups read through", async () => { test("named groups read through", async () => {
expect(await value(`const m = /(?<word>[a-z]+)-(?<num>\\d+)/.exec("id ab-42"); return m.groups.word + m.groups.num`)).toBe("ab42") expect(
await value(`const m = /(?<word>[a-z]+)-(?<num>\\d+)/.exec("id ab-42"); return m.groups.word + m.groups.num`),
).toBe("ab42")
}) })
test("global exec advances lastIndex across calls", async () => { test("global exec advances lastIndex across calls", async () => {
expect(await value(` expect(
await value(`
const r = /\\d+/g const r = /\\d+/g
const first = r.exec("a1b22c") const first = r.exec("a1b22c")
const second = r.exec("a1b22c") const second = r.exec("a1b22c")
return [first[0], second[0]] return [first[0], second[0]]
`)).toEqual(["1", "22"]) `),
).toEqual(["1", "22"])
}) })
test("string match: non-global carries index, global lists all matches", async () => { test("string match: non-global carries index, global lists all matches", async () => {
@@ -194,34 +210,51 @@ describe("RegExp", () => {
describe("Map", () => { describe("Map", () => {
test("get/set/has/size with chaining", async () => { test("get/set/has/size with chaining", async () => {
expect(await value(` expect(
await value(`
const m = new Map() const m = new Map()
m.set("a", 1).set("b", 2) m.set("a", 1).set("b", 2)
return { a: m.get("a"), b: m.get("b"), has: m.has("a"), miss: m.get("zz") === undefined, size: m.size } return { a: m.get("a"), b: m.get("b"), has: m.has("a"), miss: m.get("zz") === undefined, size: m.size }
`)).toEqual({ a: 1, b: 2, has: true, miss: true, size: 5 - 3 }) `),
).toEqual({ a: 1, b: 2, has: true, miss: true, size: 5 - 3 })
}) })
test("object keys use identity", async () => { test("object keys use identity", async () => {
expect(await value(` expect(
await value(`
const key = { id: 1 } const key = { id: 1 }
const m = new Map() const m = new Map()
m.set(key, "hit") m.set(key, "hit")
return [m.get(key), m.get({ id: 1 }) === undefined] return [m.get(key), m.get({ id: 1 }) === undefined]
`)).toEqual(["hit", true]) `),
).toEqual(["hit", true])
}) })
test("construction from entry pairs and another Map", async () => { test("construction from entry pairs and another Map", async () => {
expect(await value(`const m = new Map([["a", 1], ["b", 2]]); return m.get("b")`)).toBe(2) expect(await value(`const m = new Map([["a", 1], ["b", 2]]); return m.get("b")`)).toBe(2)
expect(await value(`const m = new Map([["a", 1]]); const n = new Map(m); n.set("b", 2); return [n.get("a"), n.get("b"), m.has("b")]`)).toEqual([1, 2, false]) expect(
await value(
`const m = new Map([["a", 1]]); const n = new Map(m); n.set("b", 2); return [n.get("a"), n.get("b"), m.has("b")]`,
),
).toEqual([1, 2, false])
expect((await error(`return new Map("nope")`)).message).toMatch(/\[key, value\] pairs/) expect((await error(`return new Map("nope")`)).message).toMatch(/\[key, value\] pairs/)
expect((await error(`return new Map(["flat"])`)).message).toMatch(/\[key, value\] pairs/) expect((await error(`return new Map(["flat"])`)).message).toMatch(/\[key, value\] pairs/)
}) })
test("keys/values/entries return arrays", async () => { test("keys/values/entries return arrays", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
return { keys: m.keys(), values: m.values(), entries: m.entries() } return { keys: m.keys(), values: m.values(), entries: m.entries() }
`)).toEqual({ keys: ["a", "b"], values: [1, 2], entries: [["a", 1], ["b", 2]] }) `),
).toEqual({
keys: ["a", "b"],
values: [1, 2],
entries: [
["a", 1],
["b", 2],
],
})
}) })
test("Object.fromEntries(map) and Array.from(map)", async () => { test("Object.fromEntries(map) and Array.from(map)", async () => {
@@ -230,13 +263,15 @@ describe("Map", () => {
}) })
test("for...of iterates [key, value] pairs with destructuring", async () => { test("for...of iterates [key, value] pairs with destructuring", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
let total = 0 let total = 0
let names = "" let names = ""
for (const [key, count] of m) { names += key; total += count } for (const [key, count] of m) { names += key; total += count }
return names + total return names + total
`)).toBe("ab3") `),
).toBe("ab3")
}) })
test("spread produces entry pairs", async () => { test("spread produces entry pairs", async () => {
@@ -244,32 +279,38 @@ describe("Map", () => {
}) })
test("forEach passes (value, key)", async () => { test("forEach passes (value, key)", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
const seen = [] const seen = []
m.forEach((count, key) => seen.push(key + count)) m.forEach((count, key) => seen.push(key + count))
return seen return seen
`)).toEqual(["a1", "b2"]) `),
).toEqual(["a1", "b2"])
}) })
test("delete and clear", async () => { test("delete and clear", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
const removed = m.delete("a") const removed = m.delete("a")
const missed = m.delete("zz") const missed = m.delete("zz")
const sizeAfterDelete = m.size const sizeAfterDelete = m.size
m.clear() m.clear()
return [removed, missed, sizeAfterDelete, m.size] return [removed, missed, sizeAfterDelete, m.size]
`)).toEqual([true, false, 1, 0]) `),
).toEqual([true, false, 1, 0])
}) })
test("counting idiom: grouped tallies", async () => { test("counting idiom: grouped tallies", async () => {
expect(await value(` expect(
await value(`
const words = ["a", "b", "a", "c", "a"] const words = ["a", "b", "a", "c", "a"]
const counts = new Map() const counts = new Map()
for (const word of words) counts.set(word, (counts.get(word) ?? 0) + 1) for (const word of words) counts.set(word, (counts.get(word) ?? 0) + 1)
return Object.fromEntries(counts) return Object.fromEntries(counts)
`)).toEqual({ a: 3, b: 1, c: 1 }) `),
).toEqual({ a: 3, b: 1, c: 1 })
}) })
test("maps serialize to {} at the boundary, like JSON", async () => { test("maps serialize to {} at the boundary, like JSON", async () => {
@@ -286,12 +327,14 @@ describe("Map", () => {
describe("Set", () => { describe("Set", () => {
test("add/has/delete/size with chaining", async () => { test("add/has/delete/size with chaining", async () => {
expect(await value(` expect(
await value(`
const s = new Set() const s = new Set()
s.add(1).add(2).add(1) s.add(1).add(2).add(1)
const removed = s.delete(2) const removed = s.delete(2)
return [s.size, s.has(1), s.has(2), removed] return [s.size, s.has(1), s.has(2), removed]
`)).toEqual([1, true, false, true]) `),
).toEqual([1, true, false, true])
}) })
test("dedupe idiom: [...new Set(items)]", async () => { test("dedupe idiom: [...new Set(items)]", async () => {
@@ -308,11 +351,13 @@ describe("Set", () => {
}) })
test("for...of iterates values", async () => { test("for...of iterates values", async () => {
expect(await value(` expect(
await value(`
let total = 0 let total = 0
for (const n of new Set([1, 2, 3])) total += n for (const n of new Set([1, 2, 3])) total += n
return total return total
`)).toBe(6) `),
).toBe(6)
}) })
test("sets serialize to {} at the boundary, like JSON", async () => { test("sets serialize to {} at the boundary, like JSON", async () => {
@@ -338,21 +383,32 @@ describe("stdlib integration", () => {
}) })
test("dates inside Map values survive in-sandbox reads", async () => { test("dates inside Map values survive in-sandbox reads", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["start", new Date(1000)]]) const m = new Map([["start", new Date(1000)]])
return m.get("start").getTime() return m.get("start").getTime()
`)).toBe(1000) `),
).toBe(1000)
}) })
test("instanceof recognizes the stdlib value types", async () => { test("instanceof recognizes the stdlib value types", async () => {
expect(await value(`return [new Date(0) instanceof Date, /a/ instanceof RegExp, new Map() instanceof Map, new Set() instanceof Set]`)).toEqual([true, true, true, true]) expect(
expect(await value(`return [[1] instanceof Array, [1] instanceof Object, ({}) instanceof Object, 5 instanceof Object]`)).toEqual([true, true, true, false]) await value(
`return [new Date(0) instanceof Date, /a/ instanceof RegExp, new Map() instanceof Map, new Set() instanceof Set]`,
),
).toEqual([true, true, true, true])
expect(
await value(`return [[1] instanceof Array, [1] instanceof Object, ({}) instanceof Object, 5 instanceof Object]`),
).toEqual([true, true, true, false])
expect(await value(`return [new Map() instanceof Set, "s" instanceof Date]`)).toEqual([false, false]) expect(await value(`return [new Map() instanceof Set, "s" instanceof Date]`)).toEqual([false, false])
expect(await value(`const p = Promise.resolve(1); const isPromise = p instanceof Promise; await p; return isPromise`)).toBe(true) expect(
await value(`const p = Promise.resolve(1); const isPromise = p instanceof Promise; await p; return isPromise`),
).toBe(true)
}) })
test("realistic pipeline: parse, extract with regex, dedupe, count by day", async () => { test("realistic pipeline: parse, extract with regex, dedupe, count by day", async () => {
expect(await value(` expect(
await value(`
const raw = '[{"at":"2024-01-01T05:00:00Z","tag":"a b"},{"at":"2024-01-01T09:00:00Z","tag":"b c"},{"at":"2024-01-02T01:00:00Z","tag":"a"}]' const raw = '[{"at":"2024-01-01T05:00:00Z","tag":"a b"},{"at":"2024-01-01T09:00:00Z","tag":"b c"},{"at":"2024-01-02T01:00:00Z","tag":"a"}]'
const rows = JSON.parse(raw) const rows = JSON.parse(raw)
const tags = new Set() const tags = new Set()
@@ -363,27 +419,34 @@ describe("stdlib integration", () => {
byDay.set(day, (byDay.get(day) ?? 0) + 1) byDay.set(day, (byDay.get(day) ?? 0) + 1)
} }
return { tags: [...tags].sort((a, b) => (a < b ? -1 : 1)), byDay: Object.fromEntries(byDay) } return { tags: [...tags].sort((a, b) => (a < b ? -1 : 1)), byDay: Object.fromEntries(byDay) }
`)).toEqual({ tags: ["a", "b", "c"], byDay: { "2024-01-01": 2, "2024-01-02": 1 } }) `),
).toEqual({ tags: ["a", "b", "c"], byDay: { "2024-01-01": 2, "2024-01-02": 1 } })
}) })
}) })
describe("sandbox values at intra-sandbox checkpoints", () => { describe("sandbox values at intra-sandbox checkpoints", () => {
test("Object.values/entries keep Dates usable", async () => { test("Object.values/entries keep Dates usable", async () => {
expect(await value(`return Object.values({ d: new Date(0) })[0].getTime()`)).toBe(0) expect(await value(`return Object.values({ d: new Date(0) })[0].getTime()`)).toBe(0)
expect(await value(`const [key, d] = Object.entries({ d: new Date(0) })[0]; return key + ":" + d.getTime()`)).toBe("d:0") expect(await value(`const [key, d] = Object.entries({ d: new Date(0) })[0]; return key + ":" + d.getTime()`)).toBe(
"d:0",
)
}) })
test("Object.assign keeps Maps usable", async () => { test("Object.assign keeps Maps usable", async () => {
expect(await value(`const merged = Object.assign({}, { m: new Map([["a", 1]]) }); return merged.m.get("a")`)).toBe(1) expect(await value(`const merged = Object.assign({}, { m: new Map([["a", 1]]) }); return merged.m.get("a")`)).toBe(
1,
)
}) })
test("object and array spread keep sandbox values usable", async () => { test("object and array spread keep sandbox values usable", async () => {
expect(await value(` expect(
await value(`
const src = { m: new Map([["a", 1]]) } const src = { m: new Map([["a", 1]]) }
const copy = { ...src } const copy = { ...src }
copy.m.set("b", 2) copy.m.set("b", 2)
return [copy.m.get("a"), src.m.get("b")] return [copy.m.get("a"), src.m.get("b")]
`)).toEqual([1, 2]) `),
).toEqual([1, 2])
expect(await value(`const list = [new Date(1000)]; const copy = [...list]; return copy[0].getTime()`)).toBe(1000) expect(await value(`const list = [new Date(1000)]; const copy = [...list]; return copy[0].getTime()`)).toBe(1000)
}) })
@@ -404,7 +467,10 @@ describe("sandbox values at intra-sandbox checkpoints", () => {
}) })
test("the host boundary still serializes JSON forms: results, JSON.stringify, and tool arguments", async () => { test("the host boundary still serializes JSON forms: results, JSON.stringify, and tool arguments", async () => {
expect(await value(`return { d: new Date(0), m: new Map([["a", 1]]) }`)).toEqual({ d: "1970-01-01T00:00:00.000Z", m: {} }) expect(await value(`return { d: new Date(0), m: new Map([["a", 1]]) }`)).toEqual({
d: "1970-01-01T00:00:00.000Z",
m: {},
})
expect(await value(`return JSON.stringify({ d: new Date(0) })`)).toBe('{"d":"1970-01-01T00:00:00.000Z"}') expect(await value(`return JSON.stringify({ d: new Date(0) })`)).toBe('{"d":"1970-01-01T00:00:00.000Z"}')
const observed: Array<unknown> = [] const observed: Array<unknown> = []
@@ -417,10 +483,12 @@ describe("sandbox values at intra-sandbox checkpoints", () => {
return "ok" return "ok"
}), }),
}) })
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
tools: { host: { capture } }, tools: { host: { capture } },
code: `return await tools.host.capture({ when: new Date(0), tags: new Map([["a", 1]]) })`, code: `return await tools.host.capture({ when: new Date(0), tags: new Map([["a", 1]]) })`,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(observed).toStrictEqual([{ when: "1970-01-01T00:00:00.000Z", tags: {} }]) expect(observed).toStrictEqual([{ when: "1970-01-01T00:00:00.000Z", tags: {} }])
}) })
+1 -4
View File
@@ -206,10 +206,7 @@ export const prepare = Effect.fn("LLMRequestPrep.prepare")(function* (input: Pre
}) })
function resolveTools(input: Pick<PrepareInput, "tools" | "agent" | "permission" | "user">) { function resolveTools(input: Pick<PrepareInput, "tools" | "agent" | "permission" | "user">) {
const visible = Permission.visibleTools( const visible = Permission.visibleTools(input.tools, Permission.merge(input.agent.permission, input.permission ?? []))
input.tools,
Permission.merge(input.agent.permission, input.permission ?? []),
)
return Record.filter(visible, (_, k) => input.user.tools?.[k] !== false) return Record.filter(visible, (_, k) => input.user.tools?.[k] !== false)
} }
+7 -3
View File
@@ -104,7 +104,8 @@ export function groupByServer(
const byLongest = [...servers].sort((a, b) => b.length - a.length) const byLongest = [...servers].sort((a, b) => b.length - a.length)
const groups = new Map<string, CatalogEntry[]>() const groups = new Map<string, CatalogEntry[]>()
for (const key of Object.keys(mcpTools).sort((a, b) => a.localeCompare(b))) { for (const key of Object.keys(mcpTools).sort((a, b) => a.localeCompare(b))) {
const server = byLongest.find((name) => key.startsWith(name + "_")) ?? (key.includes("_") ? key.slice(0, key.indexOf("_")) : key) const server =
byLongest.find((name) => key.startsWith(name + "_")) ?? (key.includes("_") ? key.slice(0, key.indexOf("_")) : key)
const local = server && key.startsWith(server + "_") ? key.slice(server.length + 1) : key const local = server && key.startsWith(server + "_") ? key.slice(server.length + 1) : key
const def = mcpDefs[key] const def = mcpDefs[key]
const entry: CatalogEntry = { const entry: CatalogEntry = {
@@ -129,7 +130,9 @@ export function buildCatalog(
mcpDefs: Record<string, MCPToolDef>, mcpDefs: Record<string, MCPToolDef>,
servers: readonly string[], servers: readonly string[],
): CatalogEntry[] { ): CatalogEntry[] {
return [...groupByServer(mcpTools, servers, mcpDefs).values()].flat().filter((entry) => entry.tool.execute !== undefined) return [...groupByServer(mcpTools, servers, mcpDefs).values()]
.flat()
.filter((entry) => entry.tool.execute !== undefined)
} }
/** /**
@@ -334,7 +337,8 @@ export const CodeModeTool = Tool.define(
const collect = (attachment: Attachment) => void attachments.push(attachment) const collect = (attachment: Attachment) => void attachments.push(attachment)
// Stream the current call list to the UI. Sent on every status change so the // Stream the current call list to the UI. Sent on every status change so the
// tool part shows each child call appearing and resolving while the program runs. // tool part shows each child call appearing and resolving while the program runs.
const publish = () => ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } }) const publish = () =>
ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } })
// One CodeMode tool per MCP tool, running the same shared middle as legacy // One CodeMode tool per MCP tool, running the same shared middle as legacy
// per-tool registration (McpInvoke.invoke: plugin before hook → permission // per-tool registration (McpInvoke.invoke: plugin before hook → permission
+4 -2
View File
@@ -276,7 +276,10 @@ const layer = Layer.effect(
// fresh per turn so it tracks live tool-list changes. Hard-denied tools (the shared // fresh per turn so it tracks live tool-list changes. Hard-denied tools (the shared
// Permission.visibleTools predicate over the agent's ruleset) never enter the // Permission.visibleTools predicate over the agent's ruleset) never enter the
// catalog, its inlined signatures, or the in-program search index. // catalog, its inlined signatures, or the in-program search index.
const describeCodeMode = Effect.fn("ToolRegistry.describeCodeMode")(function* (agent: Agent.Info, permission?: PermissionV1.Ruleset) { const describeCodeMode = Effect.fn("ToolRegistry.describeCodeMode")(function* (
agent: Agent.Info,
permission?: PermissionV1.Ruleset,
) {
const visible = Permission.visibleTools(yield* mcp.tools(), Permission.merge(agent.permission, permission ?? [])) const visible = Permission.visibleTools(yield* mcp.tools(), Permission.merge(agent.permission, permission ?? []))
const servers = Object.keys(yield* mcp.clients()).map(McpCatalog.sanitize) const servers = Object.keys(yield* mcp.clients()).map(McpCatalog.sanitize)
return catalogInstructions(visible, yield* mcp.defs(), servers) return catalogInstructions(visible, yield* mcp.defs(), servers)
@@ -341,7 +344,6 @@ const layer = Layer.effect(
}), }),
) )
function isZodType(value: unknown): value is z.ZodType { function isZodType(value: unknown): value is z.ZodType {
return typeof value === "object" && value !== null && "_zod" in value return typeof value === "object" && value !== null && "_zod" in value
} }
+1 -2
View File
@@ -96,8 +96,7 @@ function resolveTools(trigger?: Plugin.Interface["trigger"]) {
Layer.mergeAll( Layer.mergeAll(
Layer.mock(Permission.Service, { ask: () => Effect.void }), Layer.mock(Permission.Service, { ask: () => Effect.void }),
Layer.mock(Plugin.Service, { Layer.mock(Plugin.Service, {
trigger: trigger: trigger ?? (((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"]),
trigger ?? (((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"]),
}), }),
Layer.mock(Truncate.Service, { Layer.mock(Truncate.Service, {
output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), output: (text: string) => Effect.succeed({ content: text, truncated: false as const }),
@@ -21,8 +21,7 @@ import type { Tool as AITool } from "ai"
import { Effect, Layer } from "effect" import { Effect, Layer } from "effect"
// A 1x1 transparent PNG, base64-encoded, used to exercise image attachments. // A 1x1 transparent PNG, base64-encoded, used to exercise image attachments.
const PNG = const PNG = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
const SERVER = "fixtures" const SERVER = "fixtures"
@@ -163,8 +162,8 @@ async function buildTool() {
// this real in-memory server listed — the same snapshot shape the live service returns. // this real in-memory server listed — the same snapshot shape the live service returns.
const layer = Layer.mergeAll( const layer = Layer.mergeAll(
Layer.mock(Plugin.Service, { Layer.mock(Plugin.Service, {
trigger: (((_name: unknown, _input: unknown, output: unknown) => trigger: ((_name: unknown, _input: unknown, output: unknown) =>
Effect.succeed(output)) as Plugin.Interface["trigger"]), Effect.succeed(output)) as Plugin.Interface["trigger"],
}), }),
Layer.mock(Truncate.Service, { Layer.mock(Truncate.Service, {
output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), output: (text: string) => Effect.succeed({ content: text, truncated: false as const }),
@@ -196,9 +195,7 @@ describe("code mode integration (real MCP server)", () => {
test("the appended catalog inlines full signatures with real MCP schemas", () => { test("the appended catalog inlines full signatures with real MCP schemas", () => {
expect(description).toContain("Available tools (COMPLETE list") expect(description).toContain("Available tools (COMPLETE list")
expect(description).toContain("- fixtures (4 tools)") expect(description).toContain("- fixtures (4 tools)")
expect(description).toContain( expect(description).toContain("tools.fixtures.add(input: { a: number; b: number }): Promise<{ sum: number }>")
"tools.fixtures.add(input: { a: number; b: number }): Promise<{ sum: number }>",
)
expect(description).toContain("tools.fixtures.get_text(input: { name: string }): Promise<unknown>") expect(description).toContain("tools.fixtures.get_text(input: { name: string }): Promise<unknown>")
expect(description).toContain("// Add two numbers and return the structured sum") expect(description).toContain("// Add two numbers and return the structured sum")
// Small catalog: everything is inline, so no discovery tool is advertised. // Small catalog: everything is inline, so no discovery tool is advertised.
+22 -16
View File
@@ -193,7 +193,9 @@ describe("code mode execute", () => {
// never cherry-picks a catalog tool or fabricates result fields. // never cherry-picks a catalog tool or fabricates result fields.
expect(description).toContain("## Workflow") expect(description).toContain("## Workflow")
expect(description).toContain("1. Pick a tool from the list under `## Available tools`") expect(description).toContain("1. Pick a tool from the list under `## Available tools`")
expect(description).toContain('`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string') expect(description).toContain(
'`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string',
)
expect(description).toContain("Return only the fields you need") expect(description).toContain("Return only the fields you need")
expect(description).not.toContain("total_count") expect(description).not.toContain("total_count")
}) })
@@ -249,7 +251,9 @@ describe("code mode execute", () => {
expect(description).toContain("tools.$codemode.search(") expect(description).toContain("tools.$codemode.search(")
// PARTIAL catalogs put search first in the workflow and advertise namespace browsing. // PARTIAL catalogs put search first in the workflow and advertise namespace browsing.
expect(description).toContain("1. Find a tool (skip when it is already listed below)") expect(description).toContain("1. Find a tool (skip when it is already listed below)")
expect(description).toContain('- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.') expect(description).toContain(
'- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.',
)
expect(description).not.toContain("total_count") expect(description).not.toContain("total_count")
// All op lines cost the same estimated tokens (chars/4 rounds away the 1- vs 3-digit // All op lines cost the same estimated tokens (chars/4 rounds away the 1- vs 3-digit
// name difference), so the path tiebreak decides: the lexicographically-first ops made // name difference), so the path tiebreak decides: the lexicographically-first ops made
@@ -290,7 +294,10 @@ describe("code mode execute", () => {
linear_search: mcpTool("search", () => ""), linear_search: mcpTool("search", () => ""),
}) })
const output = await Effect.runPromise( const output = await Effect.runPromise(
tool.execute({ code: "const namespaces = Object.keys(tools); return { namespaces, count: namespaces.length }" }, ctx), tool.execute(
{ code: "const namespaces = Object.keys(tools); return { namespaces, count: namespaces.length }" },
ctx,
),
) )
expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear"], count: 2 }) expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear"], count: 2 })
}) })
@@ -565,9 +572,7 @@ describe("code mode execute", () => {
const tool = await build({ const tool = await build({
shot_take: mcpTool("take", () => ({ content: [{ type: "image", data: "PNGDATA", mimeType: "image/png" }] })), shot_take: mcpTool("take", () => ({ content: [{ type: "image", data: "PNGDATA", mimeType: "image/png" }] })),
}) })
const out = await Effect.runPromise( const out = await Effect.runPromise(tool.execute({ code: "await tools.shot.take({}); return 'captured'" }, ctx))
tool.execute({ code: "await tools.shot.take({}); return 'captured'" }, ctx),
)
expect(out.output).toBe("captured") expect(out.output).toBe("captured")
expect(out.attachments).toHaveLength(1) expect(out.attachments).toHaveLength(1)
}) })
@@ -690,9 +695,7 @@ describe("code mode permission visibility", () => {
expect(called).toEqual([]) expect(called).toEqual([])
// The rest of the namespace still works. // The rest of the namespace still works.
const allowed = await Effect.runPromise( const allowed = await Effect.runPromise(tool.execute({ code: "return await tools.github.list_issues({})" }, ctx))
tool.execute({ code: "return await tools.github.list_issues({})" }, ctx),
)
expect(allowed.metadata.error).toBeUndefined() expect(allowed.metadata.error).toBeUndefined()
expect(allowed.output).toBe("ok") expect(allowed.output).toBe("ok")
}) })
@@ -706,9 +709,7 @@ describe("code mode permission visibility", () => {
["github"], ["github"],
[askRule("github_list_issues")], [askRule("github_list_issues")],
) )
const out = await Effect.runPromise( const out = await Effect.runPromise(tool.execute({ code: "return await tools.github.list_issues({})" }, askCtx))
tool.execute({ code: "return await tools.github.list_issues({})" }, askCtx),
)
expect(out.output).toBe("ok") expect(out.output).toBe("ok")
expect(asked).toEqual(["github_list_issues"]) expect(asked).toEqual(["github_list_issues"])
}) })
@@ -733,16 +734,21 @@ describe("toSandboxResult", () => {
test("prefers structuredContent over text", () => { test("prefers structuredContent over text", () => {
const { collect } = collector() const { collect } = collector()
expect(toSandboxResult({ structuredContent: { x: 1 }, content: [{ type: "text", text: "hi" }] }, collect)).toEqual( expect(toSandboxResult({ structuredContent: { x: 1 }, content: [{ type: "text", text: "hi" }] }, collect)).toEqual({
{ x: 1 }, x: 1,
) })
}) })
test("joins text content when no structured content is present", () => { test("joins text content when no structured content is present", () => {
const { collect } = collector() const { collect } = collector()
expect( expect(
toSandboxResult( toSandboxResult(
{ content: [{ type: "text", text: "one" }, { type: "text", text: "two" }] }, {
content: [
{ type: "text", text: "one" },
{ type: "text", text: "two" },
],
},
collect, collect,
), ),
).toBe("one\ntwo") ).toBe("one\ntwo")
+7 -1
View File
@@ -2355,7 +2355,13 @@ function Execute(props: ToolProps) {
return ( return (
<> <>
<InlineTool <InlineTool
icon={props.part.state.status === "completed" && !hasRuntimeError() ? "✓" : props.part.state.status === "error" || hasRuntimeError() ? "✗" : "│"} icon={
props.part.state.status === "completed" && !hasRuntimeError()
? "✓"
: props.part.state.status === "error" || hasRuntimeError()
? "✗"
: "│"
}
color={hasRuntimeError() ? theme.error : undefined} color={hasRuntimeError() ? theme.error : undefined}
spinner={isLoading()} spinner={isLoading()}
pending="execute" pending="execute"