chore: generate

This commit is contained in:
opencode-agent[bot]
2026-07-03 04:49:44 +00:00
parent cb93114424
commit 83c638eaac
20 changed files with 2137 additions and 1207 deletions
+12 -11
View File
@@ -8,11 +8,11 @@ The package is currently private to this workspace. Its API is designed around t
```ts ```ts
// One execution // One execution
yield* CodeMode.execute({ tools, code }) yield * CodeMode.execute({ tools, code })
// A reusable runtime // A reusable runtime
const runtime = CodeMode.make({ tools, limits }) const runtime = CodeMode.make({ tools, limits })
yield* runtime.execute(code) yield * runtime.execute(code)
// One agent-facing code tool // One agent-facing code tool
const codeTool = runtime.agentTool() const codeTool = runtime.agentTool()
@@ -55,7 +55,9 @@ const runtime = CodeMode.make({
}, },
}) })
const result = yield* runtime.execute(` const result =
yield *
runtime.execute(`
const order = await tools.orders.lookup({ id: "order_42" }) const order = await tools.orders.lookup({ id: "order_42" })
return { id: order.id, needsAttention: order.status !== "complete" } return { id: order.id, needsAttention: order.status !== "complete" }
`) `)
@@ -89,13 +91,15 @@ The description and schemas are part of the model-visible tool contract. Keep de
Use `CodeMode.execute` for a single execution: Use `CodeMode.execute` for a single execution:
```ts ```ts
const result = yield* CodeMode.execute({ const result =
yield *
CodeMode.execute({
tools: { orders: { lookup: lookupOrder } }, tools: { orders: { lookup: lookupOrder } },
code: `return await tools.orders.lookup({ id: "order_42" })`, code: `return await tools.orders.lookup({ id: "order_42" })`,
limits: { maxToolCalls: 10 }, limits: { maxToolCalls: 10 },
onToolCallStart: (call) => Effect.logDebug("CodeMode tool started", call), onToolCallStart: (call) => Effect.logDebug("CodeMode tool started", call),
onToolCallEnd: (call) => Effect.logDebug("CodeMode tool settled", call), onToolCallEnd: (call) => Effect.logDebug("CodeMode tool settled", call),
}) })
``` ```
The Effect environment is inferred from the supplied tools. CodeMode does not erase service requirements introduced by tool implementations. The Effect environment is inferred from the supplied tools. CodeMode does not erase service requirements introduced by tool implementations.
@@ -223,7 +227,7 @@ CodeMode is an orchestration language, not a general JavaScript runtime.
The limits are exactly three knobs: The limits are exactly three knobs:
| Limit | Default | Bounds | | Limit | Default | Bounds |
| --- | ---: | --- | | ---------------- | -------------------: | -------------------------------------------------------------------- |
| `timeoutMs` | none — no timeout | Wall-clock execution time. | | `timeoutMs` | none — no timeout | Wall-clock execution time. |
| `maxToolCalls` | none — unlimited | Tool calls admitted during the execution. | | `maxToolCalls` | none — unlimited | Tool calls admitted during the execution. |
| `maxOutputBytes` | none — no truncation | Model-facing output: the serialized result value plus captured logs. | | `maxOutputBytes` | none — no truncation | Model-facing output: the serialized result value plus captured logs. |
@@ -255,7 +259,7 @@ Two interpreter internals are fixed constants rather than knobs: at most 8 tool
Failures are data: Failures are data:
| Kind | Meaning | | Kind | Meaning |
| --- | --- | | ----------------------- | -------------------------------------------------------------------------------------------------------- |
| `ParseError` | Source is empty or cannot be parsed. | | `ParseError` | Source is empty or cannot be parsed. |
| `UnsupportedSyntax` | Parsed JavaScript is outside the supported subset. | | `UnsupportedSyntax` | Parsed JavaScript is outside the supported subset. |
| `UnknownTool` | A program referenced a tool the host did not provide. | | `UnknownTool` | A program referenced a tool the host did not provide. |
@@ -272,10 +276,7 @@ Unknown host failures, defects, invalid outputs, and copying failures are saniti
```ts ```ts
import { toolError } from "@opencode-ai/codemode" import { toolError } from "@opencode-ai/codemode"
run: ({ id }) => run: ({ id }) => (authorized(id) ? loadOrder(id) : Effect.fail(toolError("Order is unavailable")))
authorized(id)
? loadOrder(id)
: Effect.fail(toolError("Order is unavailable"))
``` ```
Only the supplied message is model-visible. The optional cause is never returned in `ExecuteResult`; hosts should perform any required internal logging before crossing this boundary. Only the supplied message is model-visible. The optional cause is never returned in `ExecuteResult`; hosts should perform any required internal logging before crossing this boundary.
+96 -66
View File
@@ -39,6 +39,7 @@ package and was **deleted** in Wave 3 (done, see below).
From issue #34787 and design discussion. Do not relitigate these casually. From issue #34787 and design discussion. Do not relitigate these casually.
### Core direction ### Core direction
- Generic CodeMode lives in its own package: `@opencode-ai/codemode` (repo scope convention; - Generic CodeMode lives in its own package: `@opencode-ai/codemode` (repo scope convention;
the issue's `@opencode/codemode` name was normalized to the `@opencode-ai/*` convention). the issue's `@opencode/codemode` name was normalized to the `@opencode-ai/*` convention).
- **Keep the hand-rolled interpreter.** No QuickJS/V8/sandbox-engine dependency. We own and - **Keep the hand-rolled interpreter.** No QuickJS/V8/sandbox-engine dependency. We own and
@@ -52,6 +53,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
products/blog posts) in code, comments, commit messages, or docs in this repo. products/blog posts) in code, comments, commit messages, or docs in this repo.
### MCP / tools ### MCP / tools
- The MCP adapter lives in OpenCode, not here. It converts MCP definitions into ordinary - The MCP adapter lives in OpenCode, not here. It converts MCP definitions into ordinary
`Tool.make(...)` definitions and hands CodeMode a plain tool tree. `Tool.make(...)` definitions and hands CodeMode a plain tool tree.
- Permissions stay in the OpenCode adapter (each tool's `run` wraps the permission ask). - Permissions stay in the OpenCode adapter (each tool's `run` wraps the permission ask).
@@ -61,8 +63,9 @@ From issue #34787 and design discussion. Do not relitigate these casually.
`tools.<server>.<tool>` namespaces before handing them over. `tools.<server>.<tool>` namespaces before handing them over.
### Discovery / search ### Discovery / search
- **Search only — no separate `describe`.** `tools.$codemode.search({ query?, namespace?, - **Search only — no separate `describe`.** `tools.$codemode.search({ query?, namespace?,
limit? })` over the final tool tree, owned by this package. limit? })` over the final tool tree, owned by this package.
- Search result item shape: `{ path, description, signature }` in an `{ items, total }` - Search result item shape: `{ path, description, signature }` in an `{ items, total }`
wrapper. The `signature` string embeds the full input/output TypeScript types — in search wrapper. The `signature` string embeds the full input/output TypeScript types — in search
results it is the pretty, JSDoc-annotated multiline form (Fix 7), so per-field schema results it is the pretty, JSDoc-annotated multiline form (Fix 7), so per-field schema
@@ -82,6 +85,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
- Tools without an output schema render `unknown` as their return type. - Tools without an output schema render `unknown` as their return type.
### Schemas / Tool.make ### Schemas / Tool.make
- `Tool.make` carries rich metadata so search can render real signatures. - `Tool.make` carries rich metadata so search can render real signatures.
- Support **Effect Schema** (first-class, validating) and **JSON Schema** (initially - Support **Effect Schema** (first-class, validating) and **JSON Schema** (initially
render-only — used for TypeScript rendering; the adapter may validate on its own). Leave render-only — used for TypeScript rendering; the adapter may validate on its own). Leave
@@ -90,6 +94,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
normalization for plugin authors can come later. normalization for plugin authors can come later.
### Attachments / output ### Attachments / output
- **No `output.text/file/image` API in v1.** (Deleted in Wave 2.) - **No `output.text/file/image` API in v1.** (Deleted in Wave 2.)
- Tool calls return native structured payloads into the sandbox. Files/images emitted by - Tool calls return native structured payloads into the sandbox. Files/images emitted by
child tools **never enter the sandbox** — the OpenCode adapter strips and accumulates them child tools **never enter the sandbox** — the OpenCode adapter strips and accumulates them
@@ -100,6 +105,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
image bytes into context or drop attachments. image bytes into context or drop attachments.
### Runtime behavior ### Runtime behavior
- Limits are EXACTLY the three public knobs: `{ timeoutMs, maxToolCalls, maxOutputBytes }` — - Limits are EXACTLY the three public knobs: `{ timeoutMs, maxToolCalls, maxOutputBytes }` —
matching the original locked spec exactly. NO limit has a default (user direction, Fix 6 matching the original locked spec exactly. NO limit has a default (user direction, Fix 6
for the first two; extended to `maxOutputBytes` in the truncation-layering fix below): for the first two; extended to `maxOutputBytes` in the truncation-layering fix below):
@@ -129,7 +135,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
- `console.*` is captured into `logs` on the result; the host appends them to model-facing - `console.*` is captured into `logs` on the result; the host appends them to model-facing
output. Not a tool call; costs no tool budget. output. Not a tool call; costs no tool budget.
- Simple tool-call **start/end hooks** for nested progress: `onToolCallStart({ index, name, - Simple tool-call **start/end hooks** for nested progress: `onToolCallStart({ index, name,
input })` and `onToolCallEnd({ index, name, input, durationMs, outcome, message? })`. input })` and `onToolCallEnd({ index, name, input, durationMs, outcome, message? })`.
Interrupted calls fire no end event. No `CurrentToolCall` context service (removed in Interrupted calls fire no end event. No `CurrentToolCall` context service (removed in
Wave 2). Wave 2).
@@ -148,8 +154,9 @@ and `bun run typecheck`; from `packages/opencode`, `bun run typecheck` and
`test/tool/registry.test.ts`). `test/tool/registry.test.ts`).
### Wave 0 — scaffold (done) ### Wave 0 — scaffold (done)
- `packages/codemode` created from the experiments implementation: `src/{index,codemode,tool, - `packages/codemode` created from the experiments implementation: `src/{index,codemode,tool,
tool-error,tool-runtime}.ts`, README, AGENTS.md, tests. tool-error,tool-runtime}.ts`, README, AGENTS.md, tests.
- `package.json`: name `@opencode-ai/codemode`, deps `acorn@8.15.0`, `typescript: catalog:`, - `package.json`: name `@opencode-ai/codemode`, deps `acorn@8.15.0`, `typescript: catalog:`,
`effect: catalog:` (both repos pin effect `4.0.0-beta.83`; opencode's effect patch only `effect: catalog:` (both repos pin effect `4.0.0-beta.83`; opencode's effect patch only
touches `unstable/httpapi`, which this package doesn't use). touches `unstable/httpapi`, which this package doesn't use).
@@ -157,6 +164,7 @@ and `bun run typecheck`; from `packages/opencode`, `bun run typecheck` and
Context.Service key string renamed to `@opencode-ai/codemode/CurrentToolCall`. Context.Service key string renamed to `@opencode-ai/codemode/CurrentToolCall`.
### Wave 1a — forgiving JS semantics (done) ### Wave 1a — forgiving JS semantics (done)
Ported from the old opencode rune work; `test/parity.test.ts` (24 tests) is the acceptance Ported from the old opencode rune work; `test/parity.test.ts` (24 tests) is the acceptance
spec. The seeded interpreter was deliberately strict; these behaviors replaced that: spec. The seeded interpreter was deliberately strict; these behaviors replaced that:
@@ -173,6 +181,7 @@ spec. The seeded interpreter was deliberately strict; these behaviors replaced t
null/undefined still throws (real JS throws too). null/undefined still throws (real JS throws too).
### Wave 1b-i — stdlib value types: Date, RegExp, Map, Set (done) ### Wave 1b-i — stdlib value types: Date, RegExp, Map, Set (done)
`src/values.ts` holds `SandboxDate/SandboxRegExp/SandboxMap/SandboxSet` (own module so both `src/values.ts` holds `SandboxDate/SandboxRegExp/SandboxMap/SandboxSet` (own module so both
`codemode.ts` and `tool-runtime.ts` import without a cycle). Design: `codemode.ts` and `tool-runtime.ts` import without a cycle). Design:
@@ -202,13 +211,14 @@ spec. The seeded interpreter was deliberately strict; these behaviors replaced t
interpolation renders `/regex/` and ISO dates directly. interpolation renders `/regex/` and ISO dates directly.
### Wave 2 — API layer (done) ### Wave 2 — API layer (done)
The package's public contract, reshaped for the Wave 3 adapter. 101 tests / 0 fail after this The package's public contract, reshaped for the Wave 3 adapter. 101 tests / 0 fail after this
wave; both packages typecheck clean. wave; both packages typecheck clean.
- **`Tool.make` schema flexibility** (`src/tool.ts`): `input`/`output` each accept an Effect - **`Tool.make` schema flexibility** (`src/tool.ts`): `input`/`output` each accept an Effect
Schema (validating, decoded both directions as before) OR a raw JSON Schema document Schema (validating, decoded both directions as before) OR a raw JSON Schema document
(render-only — no validation, values pass through; rendering handles `$defs`/`definitions` (render-only — no validation, values pass through; rendering handles `$defs`/`definitions`
+ `$ref`). `output` is **optional** → signature renders `Promise<unknown>` and the host - `$ref`). `output` is **optional** → signature renders `Promise<unknown>` and the host
result is exposed as-is. Discrimination via `Schema.isSchema`. New helpers exported from result is exposed as-is. Discrimination via `Schema.isSchema`. New helpers exported from
`tool.ts`: `inputTypeScript`/`outputTypeScript`/`decodeInput`/`decodeOutput`/ `tool.ts`: `inputTypeScript`/`outputTypeScript`/`decodeInput`/`decodeOutput`/
`jsonSchemaToTypeScript`; `tool-runtime.ts` consumes them (no direct `Schema.*` use there `jsonSchemaToTypeScript`; `tool-runtime.ts` consumes them (no direct `Schema.*` use there
@@ -228,14 +238,14 @@ wave; both packages typecheck clean.
(`ToolError`/`ToolRuntimeError` message, else "Tool execution failed"). Interrupted calls (`ToolError`/`ToolRuntimeError` message, else "Tool execution failed"). Interrupted calls
fire no end event (timeout kills the whole execution anyway). fire no end event (timeout kills the whole execution anyway).
- **Limits collapse**: public `ExecutionLimits` = `{ timeoutMs?, maxToolCalls?, - **Limits collapse**: public `ExecutionLimits` = `{ timeoutMs?, maxToolCalls?,
maxOutputBytes? }` (defaults 10_000 / 100 / 32_000). This wave kept the other knobs as maxOutputBytes? }` (defaults 10_000 / 100 / 32_000). This wave kept the other knobs as
internal defaults reachable through an `@internal` `InternalExecutionLimits` type; Fix 5 internal defaults reachable through an `@internal` `InternalExecutionLimits` type; Fix 5
later deleted that type and the internal limit system entirely. later deleted that type and the internal limit system entirely.
- **`maxOutputBytes` truncation** (CodeMode-owned, never fails): applied via `boundOutput` in - **`maxOutputBytes` truncation** (CodeMode-owned, never fails): applied via `boundOutput` in
a final `Effect.map` over every result path (success/timeout/normalized failure). Oversized a final `Effect.map` over every result path (success/timeout/normalized failure). Oversized
serialized values become truncated text + ` [result truncated: N bytes exceeds the M-byte serialized values become truncated text + ` [result truncated: N bytes exceeds the M-byte
output limit; return a smaller value]`; logs keep leading lines within the remaining budget output limit; return a smaller value]`; logs keep leading lines within the remaining budget
+ `[logs truncated: showing K of N lines]`; result gains `truncated: true` (also added to - `[logs truncated: showing K of N lines]`; result gains `truncated: true` (also added to
`ExecuteResultSchema`). UTF-8-safe truncation (no split code points). (The in-sandbox `ExecuteResultSchema`). UTF-8-safe truncation (no split code points). (The in-sandbox
`maxDataBytes` check that used to throw first on oversized raw values died in Fix 5 — `maxDataBytes` check that used to throw first on oversized raw values died in Fix 5 —
truncation is now the only result-size mechanism.) truncation is now the only result-size mechanism.)
@@ -244,6 +254,7 @@ wave; both packages typecheck clean.
(`total: 1`), bypassing ranking. Tokenization/ranking/shape unchanged. (`total: 1`), bypassing ranking. Tokenization/ranking/shape unchanged.
### Wave 3 — OpenCode MCP adapter (done) ### Wave 3 — OpenCode MCP adapter (done)
`packages/opencode/src/session/code-mode.ts` rewritten as a thin adapter over this package; `packages/opencode/src/session/code-mode.ts` rewritten as a thin adapter over this package;
the vendored rune interpreter is gone. Same `define(mcpTools, mcpDefs, servers)` signature, so the vendored rune interpreter is gone. Same `define(mcpTools, mcpDefs, servers)` signature, so
`tools.ts` gating (flag on + MCP tools exist → single `execute` tool, early-return suppresses `tools.ts` gating (flag on + MCP tools exist → single `execute` tool, early-return suppresses
@@ -257,7 +268,7 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
invoked) — so signature rendering, the inline-vs-search switch, and `$codemode.search` invoked) — so signature rendering, the inline-vs-search switch, and `$codemode.search`
availability all come from this package and stay consistent with execution. availability all come from this package and stay consistent with execution.
- **`run` path**: per-child permission ask first (`ctx.ask({ permission: entry.key, patterns: - **`run` path**: per-child permission ask first (`ctx.ask({ permission: entry.key, patterns:
["*"], always: ["*"] })`, exactly the old gating; approving `execute` approves no child). ["*"], always: ["*"] })`, exactly the old gating; approving `execute` approves no child).
Denials and host failures are mapped to `toolError(message)` so they surface as safe, Denials and host failures are mapped to `toolError(message)` so they surface as safe,
catchable in-program failures (MCP `isError` text propagates as `e.message`; without this catchable in-program failures (MCP `isError` text propagates as `e.message`; without this
they'd be sanitized to "Tool execution failed"). Dispatch reuses the ai-sdk wrapper from they'd be sanitized to "Tool execution failed"). Dispatch reuses the ai-sdk wrapper from
@@ -270,7 +281,7 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
No handles, no `Result<T>` envelope, no base64 in the sandbox, no data-size tuning (the No handles, no `Result<T>` envelope, no base64 in the sandbox, no data-size tuning (the
`maxDataBytes` budget that existed at the time was deleted in Fix 5). `maxDataBytes` budget that existed at the time was deleted in Fix 5).
- **Execute result**: `{ output: formatValue(value) + trailing "Logs:" section (success AND - **Execute result**: `{ output: formatValue(value) + trailing "Logs:" section (success AND
error — logs are plain pre-formatted lines now), attachments: accumulated }` through the error — logs are plain pre-formatted lines now), attachments: accumulated }` through the
existing `Tool.ExecuteResult.attachments` → `message-v2.ts` vision plumbing; attachments existing `Tool.ExecuteResult.attachments` → `message-v2.ts` vision plumbing; attachments
ride on both success and error results. Diagnostic `suggestions` not already contained in ride on both success and error results. Diagnostic `suggestions` not already contained in
the message are appended to error output. Native outer truncation stays on (adapter never the message are appended to error output. Native outer truncation stays on (adapter never
@@ -296,6 +307,7 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
describe/`renderType`/`rankTools` tests died with the old design (58+17+24 → 34+16). describe/`renderType`/`rankTools` tests died with the old design (58+17+24 → 34+16).
### Wave 4 — instructions/prompting + polish (done) ### Wave 4 — instructions/prompting + polish (done)
Instructions are now the budgeted-catalog + prompting-guidance form; verified e2e against a Instructions are now the budgeted-catalog + prompting-guidance form; verified e2e against a
real MCP config. Package still 101 tests / 0 fail; opencode adapter suites still 34 + 16; both real MCP config. Package still 101 tests / 0 fail; opencode adapter suites still 34 + 16; both
packages typecheck clean. packages typecheck clean.
@@ -321,7 +333,7 @@ packages typecheck clean.
exported); `CodeMode.execute` (one-shot) passes it too, preserving the exported); `CodeMode.execute` (one-shot) passes it too, preserving the
`execute`≡`make().execute` law. A speculative `tools.$codemode.search` call on a small `execute`≡`make().execute` law. A speculative `tools.$codemode.search` call on a small
catalog now succeeds instead of `UnknownTool`, and unknown-tool suggestions always point at catalog now succeeds instead of `UnknownTool`, and unknown-tool suggestions always point at
search. Search is *advertised* in the instructions only when the inlined list is PARTIAL, search. Search is _advertised_ in the instructions only when the inlined list is PARTIAL,
keeping small-catalog instructions tight. keeping small-catalog instructions tight.
- **Prompting content** in `instructions()`, mapping 1:1 to the §5 transcript failures: - **Prompting content** in `instructions()`, mapping 1:1 to the §5 transcript failures:
parse-string-results-as-JSON, return-small, console-for-intermediates, and parse-string-results-as-JSON, return-small, console-for-intermediates, and
@@ -340,7 +352,7 @@ packages typecheck clean.
- **E2E (verified, headless)**: from the repo root with `OPENCODE_EXPERIMENTAL_CODE_MODE=1`, - **E2E (verified, headless)**: from the repo root with `OPENCODE_EXPERIMENTAL_CODE_MODE=1`,
the scratch `.opencode/opencode.jsonc` (context7, github, playwright, sentry, memory, the scratch `.opencode/opencode.jsonc` (context7, github, playwright, sentry, memory,
sequential-thinking; left uncommitted/as-is), and `bun packages/opencode/src/index.ts run sequential-thinking; left uncommitted/as-is), and `bun packages/opencode/src/index.ts run
--dangerously-skip-permissions -m opencode/claude-sonnet-4-5 "..."`. Confirmed: a single --dangerously-skip-permissions -m opencode/claude-sonnet-4-5 "..."`. Confirmed: a single
`execute` tool registered alongside core tools (per-MCP registration suppressed; MCP `execute` tool registered alongside core tools (per-MCP registration suppressed; MCP
resource tools unaffected); the live description read back as "Available tools (PARTIAL — resource tools unaffected); the live description read back as "Available tools (PARTIAL —
56 of 88 shown; find the rest with tools.$codemode.search):" with correct per-namespace 56 of 88 shown; find the rest with tools.$codemode.search):" with correct per-namespace
@@ -352,6 +364,7 @@ packages typecheck clean.
images, output truncation. images, output truncation.
### Wave 5 — Promise generalization (done) ### Wave 5 — Promise generalization (done)
First-class promise values in the interpreter; the direct-tool-call-only `Promise.all` First-class promise values in the interpreter; the direct-tool-call-only `Promise.all`
restriction (and its bespoke AST checks) is gone. Package suite is 136 tests / 0 fail (35 new restriction (and its bespoke AST checks) is gone. Package suite is 136 tests / 0 fail (35 new
in `test/promise.test.ts`); adapter suites and both typechecks unchanged/green; the opencode in `test/promise.test.ts`); adapter suites and both typechecks unchanged/green; the opencode
@@ -495,7 +508,7 @@ adapter needed **no changes**.
`tools.*`." (the second line drops the tools clause when the tree is empty). `tools.*`." (the second line drops the tools clause when the tree is empty).
- **`## Workflow`**: numbered steps — find a tool via `tools.$codemode.search` → read - **`## Workflow`**: numbered steps — find a tool via `tools.$codemode.search` → read
the `{ path, description, signature }` matches → call by path → `typeof res === the `{ path, description, signature }` matches → call by path → `typeof res ===
"string" ? JSON.parse(res) : res` → return only the needed fields. When the catalog is "string" ? JSON.parse(res) : res` → return only the needed fields. When the catalog is
COMPLETE the search/read steps collapse into "Pick a tool from the list under COMPLETE the search/read steps collapse into "Pick a tool from the list under
`## Available tools`" and the steps renumber (4 instead of 5). `## Available tools`" and the steps renumber (4 instead of 5).
- **`## Rules`**: call-by-exact-path; TEXT-is-JSON → JSON.parse; return small (never raw - **`## Rules`**: call-by-exact-path; TEXT-is-JSON → JSON.parse; return small (never raw
@@ -526,16 +539,17 @@ adapter needed **no changes**.
**Fix 4 — token-budgeted catalog (was bytes)** (user direction: signatures need a token **Fix 4 — token-budgeted catalog (was bytes)** (user direction: signatures need a token
budget; namespaces must always be present): budget; namespaces must always be present):
- `src/token.ts` added: copy of `@opencode-ai/core/util/token` (`round(chars / 4)`), so
- `src/token.ts` added: copy of `@opencode-ai/core/util/token` (`round(chars / 4)`), so
the package stays dependency-free; keep in sync if the core heuristic changes. the package stays dependency-free; keep in sync if the core heuristic changes.
- `DiscoveryOptions.maxInlineCatalogBytes` → `maxInlineCatalogTokens` (default 4,000 - `DiscoveryOptions.maxInlineCatalogBytes` → `maxInlineCatalogTokens` (default 4,000
estimated tokens ≈ the old 16,000 bytes at 4 chars/token — behavior parity, not a size estimated tokens ≈ the old 16,000 bytes at 4 chars/token — behavior parity, not a size
reduction). `discoveryPlan` charges `estimate(catalogLine(tool))` per line; cheapest-first reduction). `discoveryPlan` charges `estimate(catalogLine(tool))` per line; cheapest-first
+ stop-on-first-miss unchanged at the time (stop-on-first-miss replaced by round-robin in - stop-on-first-miss unchanged at the time (stop-on-first-miss replaced by round-robin in
Fix 8). Namespace stub lines were and remain unbudgeted — every Fix 8). Namespace stub lines were and remain unbudgeted — every
namespace always appears with its tool count, even at budget 0 (asserted in package and namespace always appears with its tool count, even at budget 0 (asserted in package and
adapter tests). adapter tests).
- Ripple: chars/4 rounding erases small line-length differences, so equal-cost lines fall - Ripple: chars/4 rounding erases small line-length differences, so equal-cost lines fall
to the lexicographic path tiebreak; the adapter's PARTIAL test now asserts the to the lexicographic path tiebreak; the adapter's PARTIAL test now asserts the
lexicographic tail (`op_99`) is excluded instead of `op_149`. Fixed-prose measurements lexicographic tail (`op_99`) is excluded instead of `op_149`. Fixed-prose measurements
(2026-07): preamble ~44 + Workflow ~146 + Rules ~362 + Syntax ~453 ≈ 1,100 tokens fixed; (2026-07): preamble ~44 + Workflow ~146 + Rules ~362 + Syntax ~453 ≈ 1,100 tokens fixed;
@@ -543,13 +557,14 @@ budget; namespaces must always be present):
**Fix 5 — internal limits removed** (user direction: only the three PUBLIC limits survive as **Fix 5 — internal limits removed** (user direction: only the three PUBLIC limits survive as
configurable knobs; the internal limit system dies): configurable knobs; the internal limit system dies):
- `ExecutionLimits` (`timeoutMs` 10_000 / `maxToolCalls` 100 / `maxOutputBytes` 32_000 at
- `ExecutionLimits` (`timeoutMs` 10_000 / `maxToolCalls` 100 / `maxOutputBytes` 32_000 at
the time; Fix 6 later removed the first two defaults. Same validation: safe integers, the time; Fix 6 later removed the first two defaults. Same validation: safe integers,
timeoutMs >= 1, others >= 0, RangeError otherwise) is now timeoutMs >= 1, others >= 0, RangeError otherwise) is now
the ENTIRE limit surface — exactly the shape §2's original locked spec named. the ENTIRE limit surface — exactly the shape §2's original locked spec named.
`ResolvedExecutionLimits` shrank to those three fields; the `@internal` `ResolvedExecutionLimits` shrank to those three fields; the `@internal`
`InternalExecutionLimits` type is deleted. `InternalExecutionLimits` type is deleted.
- **Deleted outright**: `maxOperations` and the whole operation-budget machinery - **Deleted outright**: `maxOperations` and the whole operation-budget machinery
(`recordWork`/`recordOperation`/`budget.operations`, plus the `workUnits`/ (`recordWork`/`recordOperation`/`budget.operations`, plus the `workUnits`/
`cheapArrayMethods` cost helpers); `maxSourceBytes` (the pre-parse source-size check); `cheapArrayMethods` cost helpers); `maxSourceBytes` (the pre-parse source-size check);
`maxDataBytes` (every byte-accounting path: `runtimeValueBytes`, `boundedProgramValue`, `maxDataBytes` (every byte-accounting path: `runtimeValueBytes`, `boundedProgramValue`,
@@ -561,29 +576,29 @@ configurable knobs; the internal limit system dies):
actively harmful: an MCP tool returning 20k rows failed). The `OperationLimitExceeded` actively harmful: an MCP tool returning 20k rows failed). The `OperationLimitExceeded`
and `AuditLimitExceeded` diagnostic kinds are gone from the `DiagnosticKind` union and and `AuditLimitExceeded` diagnostic kinds are gone from the `DiagnosticKind` union and
`ExecuteResultSchema` (fine — the package is unreleased). `ExecuteResultSchema` (fine — the package is unreleased).
- **Fixed constants, not knobs**: `TOOL_CALL_CONCURRENCY = 8` (codemode.ts; the fork - **Fixed constants, not knobs**: `TOOL_CALL_CONCURRENCY = 8` (codemode.ts; the fork
semaphore) and `MAX_VALUE_DEPTH = 32` (tool-runtime.ts; the `copyIn` depth check — kept semaphore) and `MAX_VALUE_DEPTH = 32` (tool-runtime.ts; the `copyIn` depth check — kept
only because it produces a clearer error than a native stack-overflow RangeError; still only because it produces a clearer error than a native stack-overflow RangeError; still
`InvalidDataValue`). The `DataLimits` plumbing through `tool-runtime.ts` is gone — `InvalidDataValue`). The `DataLimits` plumbing through `tool-runtime.ts` is gone —
`copyIn(value, label)` needs no limits argument, and `ToolRuntime.make` takes just `copyIn(value, label)` needs no limits argument, and `ToolRuntime.make` takes just
`(tools, maxToolCalls, hooks?, searchIndex?)`. `(tools, maxToolCalls, hooks?, searchIndex?)`.
- **Verified fact**: timeout interruption does NOT depend on the operation budget — the - **Verified fact**: timeout interruption does NOT depend on the operation budget — the
Effect fiber runtime auto-yields between interpreter steps, so `timeoutMs` interrupts Effect fiber runtime auto-yields between interpreter steps, so `timeoutMs` interrupts
even a pure `while (true) {}` loop (empirically verified: a 200ms timeout fired at even a pure `while (true) {}` loop (empirically verified: a 200ms timeout fired at
~225ms with maxOperations set to MAX_SAFE_INTEGER before the deletion). A regression ~225ms with maxOperations set to MAX_SAFE_INTEGER before the deletion). A regression
test in `codemode.test.ts` asserts exactly this (`while(true){}` + `timeoutMs: 200` → test in `codemode.test.ts` asserts exactly this (`while(true){}` + `timeoutMs: 200` →
`TimeoutExceeded`, elapsed well under a few seconds). `TimeoutExceeded`, elapsed well under a few seconds).
- **Kept (correctness, not budgets)**: circular detection (`copyIn` walks + - **Kept (correctness, not budgets)**: circular detection (`copyIn` walks +
`rejectCircularInsertion` on mutations), plain-objects-only, blocked properties `rejectCircularInsertion` on mutations), plain-objects-only, blocked properties
(`__proto__`/`constructor`/`prototype`), data-only checks, and all three public-limit (`__proto__`/`constructor`/`prototype`), data-only checks, and all three public-limit
behaviors unchanged. behaviors unchanged.
- Behavior deltas beyond the intended kills: in-sandbox structures deeper than 32 levels - Behavior deltas beyond the intended kills: in-sandbox structures deeper than 32 levels
now fail at the data boundary (`copyIn`) instead of at construction; array index now fail at the data boundary (`copyIn`) instead of at construction; array index
assignment allows any non-negative integer index (holes permitted, message now "must be assignment allows any non-negative integer index (holes permitted, message now "must be
a non-negative integer"); interpreter-produced deep/hostile structures that overflow the a non-negative integer"); interpreter-produced deep/hostile structures that overflow the
native stack during a walk still normalize to the existing "Execution exceeded the native stack during a walk still normalize to the existing "Execution exceeded the
maximum nesting depth." data diagnostic — failures remain data everywhere. maximum nesting depth." data diagnostic — failures remain data everywhere.
- Tests: deleted the knob-only tests (stdlib Map/Set collection-length growth ×2, - Tests: deleted the knob-only tests (stdlib Map/Set collection-length growth ×2,
enumeration operation-budget, codemode maxDataBytes/maxSourceBytes/maxOperations/ enumeration operation-budget, codemode maxDataBytes/maxSourceBytes/maxOperations/
maxConcurrency-RangeError assertions, and the adapter's runaway-loop-via-operation-limit maxConcurrency-RangeError assertions, and the adapter's runaway-loop-via-operation-limit
test — superseded by the package timeout regression test); rewrote the helpers that used test — superseded by the package timeout regression test); rewrote the helpers that used
@@ -639,7 +654,8 @@ adapter suites: 34 + 16.
**Fix 8 — condensed instructions + round-robin catalog fairness + plural-aware search** **Fix 8 — condensed instructions + round-robin catalog fairness + plural-aware search**
(user direction: the fixed instruction prose was too verbose; two discovery fixes ride (user direction: the fixed instruction prose was too verbose; two discovery fixes ride
along). All in `tool-runtime.ts`; no interpreter changes. along). All in `tool-runtime.ts`; no interpreter changes.
- **Syntax section inverted**: the three dense allowlist lines (~453 estimated tokens)
- **Syntax section inverted**: the three dense allowlist lines (~453 estimated tokens)
are replaced by four short lines (~188) built on "models already know JavaScript; name are replaced by four short lines (~188) built on "models already know JavaScript; name
only what is unusual or missing": (1) standard modern JS works — functions/closures, only what is unusual or missing": (1) standard modern JS works — functions/closures,
destructuring, template literals, loops, try/catch, spread, optional chaining, the destructuring, template literals, loops, try/catch, spread, optional chaining, the
@@ -656,11 +672,11 @@ along). All in `tool-runtime.ts`; no interpreter changes.
interfaces/type aliases are stripped and TS **enums actually work** (transpileModule interfaces/type aliases are stripped and TS **enums actually work** (transpileModule
compiles them to an IIFE the interpreter runs), hence enums deliberately unmentioned. compiles them to an IIFE the interpreter runs), hence enums deliberately unmentioned.
`supportedSyntaxMessage` (the in-diagnostic text in `codemode.ts`) is untouched. `supportedSyntaxMessage` (the in-diagnostic text in `codemode.ts`) is untouched.
- **Workflow/Rules deduped**: the call-by-exact-path, JSON.parse-string-results, and - **Workflow/Rules deduped**: the call-by-exact-path, JSON.parse-string-results, and
return-small content now lives ONLY in the numbered Workflow steps (with their return-small content now lives ONLY in the numbered Workflow steps (with their
compliance-driving justifications inline: "most tools return JSON as a string", "raw compliance-driving justifications inline: "most tools return JSON as a string", "raw
payloads get truncated and waste context"); Rules keeps only bullets adding new payloads get truncated and waste context"); Rules keeps only bullets adding new
content — filter/aggregate collections in code, console.* intermediates (logs ride content — filter/aggregate collections in code, console.\* intermediates (logs ride
back), Promise.all parallelism, Object.keys/for...in enumeration, browse-namespace back), Promise.all parallelism, Object.keys/for...in enumeration, browse-namespace
(PARTIAL only), and the media rule compressed to one line. The no-.then/.catch (PARTIAL only), and the media rule compressed to one line. The no-.then/.catch
guidance moved to the Syntax not-supported line. Content upgrades: the PARTIAL search guidance moved to the Syntax not-supported line. Content upgrades: the PARTIAL search
@@ -668,13 +684,13 @@ along). All in `tool-runtime.ts`; no interpreter changes.
clearly-a-query-string example, not a tool name), and the exact-path guidance is now clearly-a-query-string example, not a tool name), and the exact-path guidance is now
"call it with the result's `path` as-is (never guess segments)" / COMPLETE: "use it "call it with the result's `path` as-is (never guess segments)" / COMPLETE: "use it
as-is rather than guessing segments". as-is rather than guessing segments".
- **Fixed-prose measurements** (instructions split on `"\n## "`, catalog budget 0, - **Fixed-prose measurements** (instructions split on `"\n## "`, catalog budget 0,
bytes/3.7 — same method as Fix 4; chars/4 in parentheses): bytes/3.7 — same method as Fix 4; chars/4 in parentheses):
preamble 44 → 44 (41 → 41), Workflow 146 → 187 (135 → 171), Rules 362 → 191 preamble 44 → 44 (41 → 41), Workflow 146 → 187 (135 → 171), Rules 362 → 191
(332 → 176), Syntax 453 → 188 (419 → 174); fixed prose total 1,005 → 610 (927 → 562), (332 → 176), Syntax 453 → 188 (419 → 174); fixed prose total 1,005 → 610 (927 → 562),
≈ 40% reduction with no behavioral content dropped. Workflow grew slightly because it ≈ 40% reduction with no behavioral content dropped. Workflow grew slightly because it
absorbed the deduped parse/return-small justifications. absorbed the deduped parse/return-small justifications.
- **Round-robin namespace inlining** (`discoveryPlan`): the ported stop-on-first-miss - **Round-robin namespace inlining** (`discoveryPlan`): the ported stop-on-first-miss
behavior (alphabetically-late namespaces starved to "none shown" while an early behavior (alphabetically-late namespaces starved to "none shown" while an early
namespace inlines everything) is replaced by round-robin fairness — in each round namespace inlines everything) is replaced by round-robin fairness — in each round
(namespaces alphabetical), every namespace still holding un-inlined tools attempts to (namespaces alphabetical), every namespace still holding un-inlined tools attempts to
@@ -685,14 +701,14 @@ along). All in `tool-runtime.ts`; no interpreter changes.
`(N tools)`/`(N tools, K shown)`/`(N tools, none shown)` labels, COMPLETE vs PARTIAL `(N tools)`/`(N tools, K shown)`/`(N tools, none shown)` labels, COMPLETE vs PARTIAL
header, alphabetical namespace order in the output, cheapest-first within each header, alphabetical namespace order in the output, cheapest-first within each
namespace's shown set. namespace's shown set.
- **Plural/singular search fix**: `tokenize`d terms matched one-directionally (term must - **Plural/singular search fix**: `tokenize`d terms matched one-directionally (term must
be substring of indexed text), so query "issues" missed a tool whose text only says be substring of indexed text), so query "issues" missed a tool whose text only says
"issue". Now each term expands to `termForms` — the term plus naive singular variants "issue". Now each term expands to `termForms` — the term plus naive singular variants
(trailing "es" stripped when length > 3, trailing "s" when length > 2) — and each of (trailing "es" stripped when length > 3, trailing "s" when length > 2) — and each of
the four field checks passes when ANY form matches. Weights, exact-path lookup, and the four field checks passes when ANY form matches. Weights, exact-path lookup, and
namespace scoping untouched. A true plural path match still outranks a singular-only namespace scoping untouched. A true plural path match still outranks a singular-only
description match (path substring 8 + searchable 2 > description 4 + searchable 2). description match (path substring 8 + searchable 2 > description 4 + searchable 2).
- **Tests**: package instruction/structure assertions updated to the new text; new - **Tests**: package instruction/structure assertions updated to the new text; new
syntax-section test (leads with "Standard modern JavaScript works", names the syntax-section test (leads with "Standard modern JavaScript works", names the
verified not-supported list, keeps the data-boundary note); the budget-exhaustion verified not-supported list, keeps the data-boundary note); the budget-exhaustion
test rewritten to assert the new fairness (alpha.expensive not fitting must NOT test rewritten to assert the new fairness (alpha.expensive not fitting must NOT
@@ -707,25 +723,27 @@ along). All in `tool-runtime.ts`; no interpreter changes.
**Fix 9 — prompting trims per user review of Fix 8** (user reviewed the condensed **Fix 9 — prompting trims per user review of Fix 8** (user reviewed the condensed
instructions and directed further cuts): instructions and directed further cuts):
- Default `maxInlineCatalogTokens` 4,000 → **2,000** (user wants ~2k tokens of signatures
- Default `maxInlineCatalogTokens` 4,000 → **2,000** (user wants ~2k tokens of signatures
auto-inlined; round-robin fairness from Fix 8 spreads it across all namespaces). auto-inlined; round-robin fairness from Fix 8 spreads it across all namespaces).
- Console rule and files/images rule DROPPED from `## Rules`. Replaced by a single - Console rule and files/images rule DROPPED from `## Rules`. Replaced by a single
`unknown`-treatment warning: "A result typed `Promise<unknown>` has no guaranteed `unknown`-treatment warning: "A result typed `Promise<unknown>` has no guaranteed
shape — verify what actually came back before relying on its fields." (Deliberately shape — verify what actually came back before relying on its fields." (Deliberately
does NOT suggest console.log — user review: naming it there nudges models to log AND does NOT suggest console.log — user review: naming it there nudges models to log AND
return the same data; the prompt stays console-neutral, neither for nor against.) return the same data; the prompt stays console-neutral, neither for nor against.)
The media-stripping MECHANISM is unchanged and still tested; only the prose about it The media-stripping MECHANISM is unchanged and still tested; only the prose about it
is gone — the `[N images attached]` marker is self-explanatory in context. is gone — the `[N images attached]` marker is self-explanatory in context.
- Kept as-is per user: the JSON.parse workflow step (maps to the original motivating - Kept as-is per user: the JSON.parse workflow step (maps to the original motivating
transcript failure; NOT copied from prior art — see §5 note), the browse-namespace rule transcript failure; NOT copied from prior art — see §5 note), the browse-namespace rule
(undecided), no no-fetch/ambient-authority rule added (proposed, not approved). (undecided), no no-fetch/ambient-authority rule added (proposed, not approved).
- Explicitly REJECTED for now: auto-parsing JSON-looking text results at the adapter - Explicitly REJECTED for now: auto-parsing JSON-looking text results at the adapter
boundary ("could get weird" — type flips, program-sees vs tool-sent divergence). Logged boundary ("could get weird" — type flips, program-sees vs tool-sent divergence). Logged
as a next-iteration follow-up below. as a next-iteration follow-up below.
**DSL-expansion pass — interpreter-surface batch from §4** (the deferred medium-tier JS **DSL-expansion pass — interpreter-surface batch from §4** (the deferred medium-tier JS
parity items, done as one focused pass; no public API or limit changes): parity items, done as one focused pass; no public API or limit changes):
- **`instanceof` + real Error values**: the `errorConstructors` names (`Error`,
- **`instanceof` + real Error values**: the `errorConstructors` names (`Error`,
`TypeError`, `RangeError`, `SyntaxError`, `ReferenceError`, `EvalError`, `URIError`) are `TypeError`, `RangeError`, `SyntaxError`, `ReferenceError`, `EvalError`, `URIError`) are
bound globals (`ErrorConstructorReference`, callable with or without `new`; `typeof` → bound globals (`ErrorConstructorReference`, callable with or without `new`; `typeof` →
`"function"`). Error values stay the same plain `{ name, message }` null-prototype `"function"`). Error values stay the same plain `{ name, message }` null-prototype
@@ -752,25 +770,25 @@ parity items, done as one focused pass; no public API or limit changes):
`Array`, `Object` (any object/function-ish value), `Promise` (`SandboxPromise`), and `Array`, `Object` (any object/function-ish value), `Promise` (`SandboxPromise`), and
`Number`/`String`/`Boolean` (always false — no boxed values exist); anything else is a `Number`/`String`/`Boolean` (always false — no boxed values exist); anything else is a
catchable error naming the recognized constructors. catchable error naming the recognized constructors.
- **Array methods**: `splice` (mutating, returns the removed elements; insertions run - **Array methods**: `splice` (mutating, returns the removed elements; insertions run
`rejectCircularInsertion` like push/unshift; one-arg form removes to the end, undefined `rejectCircularInsertion` like push/unshift; one-arg form removes to the end, undefined
delete count removes nothing), `fill` (circular-checked value) and `copyWithin` delete count removes nothing), `fill` (circular-checked value) and `copyWithin`
(host-delegated), and `keys`/`values`/`entries` returning **arrays** (the Map/Set (host-delegated), and `keys`/`values`/`entries` returning **arrays** (the Map/Set
convention — for...of and spread work either way). The `retryableArrayMethods` convention — for...of and spread work either way). The `retryableArrayMethods`
"rewrite using map/filter" hint set emptied out and was deleted with its branch; unknown "rewrite using map/filter" hint set emptied out and was deleted with its branch; unknown
array properties still read `undefined`. array properties still read `undefined`.
- **String methods**: `localeCompare(that)` (locale/options arguments ignored — host - **String methods**: `localeCompare(that)` (locale/options arguments ignored — host
default locale; the dominant use is a sort comparator), `normalize(form?)` (invalid form default locale; the dominant use is a sort comparator), `normalize(form?)` (invalid form
→ catchable error naming the four valid forms), `trimLeft`/`trimRight` as → catchable error naming the four valid forms), `trimLeft`/`trimRight` as
trimStart/trimEnd aliases. trimStart/trimEnd aliases.
- **Actionable regex failures**: `toHostRegex` and `constructRegExp` now show the - **Actionable regex failures**: `toHostRegex` and `constructRegExp` now show the
offending pattern (or flags) plus the engine reason (deduped "Invalid regular offending pattern (or flags) plus the engine reason (deduped "Invalid regular
expression:" prefix via `regexFailureReason`) and a shared escaping hint expression:" prefix via `regexFailureReason`) and a shared escaping hint
(`escapeRegexHint`); flags failures list the valid flag letters; the (`escapeRegexHint`); flags failures list the valid flag letters; the
replaceAll/matchAll missing-`g` errors spell out the exact `/pattern/g` to write and replaceAll/matchAll missing-`g` errors spell out the exact `/pattern/g` to write and
the single-match alternative. the single-match alternative.
- **copyIn split (the important one)**: `copyIn(value, label, preserveSandboxValues = - **copyIn split (the important one)**: `copyIn(value, label, preserveSandboxValues =
false)` — recursion moved to a private `copyBounded`; `boundedData` (every intra-sandbox false)` — recursion moved to a private `copyBounded`; `boundedData` (every intra-sandbox
checkpoint: `Object.*` helpers, coercion/Array.from/join inputs, template checkpoint: `Object.*` helpers, coercion/Array.from/join inputs, template
interpolation, expression-result checkpoints) is now `copyIn(value, label, true)`, interpolation, expression-result checkpoints) is now `copyIn(value, label, true)`,
which passes `SandboxDate`/`SandboxRegExp`/`SandboxMap`/`SandboxSet` through **by which passes `SandboxDate`/`SandboxRegExp`/`SandboxMap`/`SandboxSet` through **by
@@ -787,7 +805,7 @@ parity items, done as one focused pass; no public API or limit changes):
there), so interpreter internals (`.map`/`.time`/`.regex`) can never leak; the there), so interpreter internals (`.map`/`.time`/`.regex`) can never leak; the
template-literal sandbox carve-out collapsed into `boundedData`. Object/array spread template-literal sandbox carve-out collapsed into `boundedData`. Object/array spread
already preserved instances (reference copies, no checkpoint) — now tested. already preserved instances (reference copies, no checkpoint) — now tested.
- **Console formatting**: `formatConsoleArgument` is total and deep - **Console formatting**: `formatConsoleArgument` is total and deep
(`formatConsoleValue`): numbers render via `String` (`NaN`/`Infinity`/`-Infinity` (`formatConsoleValue`): numbers render via `String` (`NaN`/`Infinity`/`-Infinity`
literally — never the JSON `null`; finite numbers match their JSON form), nested literally — never the JSON `null`; finite numbers match their JSON form), nested
strings are JSON-quoted, sandbox values keep their friendly forms at ANY depth (ISO strings are JSON-quoted, sandbox values keep their friendly forms at ANY depth (ISO
@@ -798,12 +816,12 @@ parity items, done as one focused pass; no public API or limit changes):
degrades to `` — console can no longer fail a program. `console.table` guards with degrades to `` — console can no longer fail a program. `console.table` guards with
`containsOpaqueReference` (sandbox cells render, e.g. ISO dates) and its row/cell `containsOpaqueReference` (sandbox cells render, e.g. ISO dates) and its row/cell
walkers treat sandbox values as scalar cells. walkers treat sandbox values as scalar cells.
- **Prose**: the instructions Syntax not-supported line dropped its `instanceof - **Prose**: the instructions Syntax not-supported line dropped its `instanceof
Error`/splice mentions (nothing else reworded); README updated (checkpoint Error`/splice mentions (nothing else reworded); README updated (checkpoint
preservation vs boundary serialization, error values/`instanceof`, new array/string preservation vs boundary serialization, error values/`instanceof`, new array/string
methods, regex-failure behavior); `supportedSyntaxMessage` left untouched (it lists methods, regex-failure behavior); `supportedSyntaxMessage` left untouched (it lists
supported syntax, was already non-exhaustive, and stays accurate). supported syntax, was already non-exhaustive, and stays accurate).
- **Tests**: package suite 169 → 209 (parity: Error/instanceof + real-JS error-name - **Tests**: package suite 169 → 209 (parity: Error/instanceof + real-JS error-name
coverage, splice/fill/copyWithin/keys/values/entries, localeCompare/normalize/trim-alias coverage, splice/fill/copyWithin/keys/values/entries, localeCompare/normalize/trim-alias
describes; stdlib: checkpoint survival incl. tool-arg boundary pinning, stdlib describes; stdlib: checkpoint survival incl. tool-arg boundary pinning, stdlib
`instanceof`, regex-message assertions; codemode: NaN/Infinity + nested/cyclic console `instanceof`, regex-message assertions; codemode: NaN/Infinity + nested/cyclic console
@@ -812,19 +830,20 @@ parity items, done as one focused pass; no public API or limit changes):
**Truncation layering — CodeMode truncation off in OpenCode** (user direction; resolves the **Truncation layering — CodeMode truncation off in OpenCode** (user direction; resolves the
§4 outer-truncation item the OPPOSITE way from "kill the outer one"): §4 outer-truncation item the OPPOSITE way from "kill the outer one"):
- `maxOutputBytes` lost its 32,000 default and now behaves exactly like the other two
- `maxOutputBytes` lost its 32,000 default and now behaves exactly like the other two
limits: absent = no truncation. All three limits are uniformly no-default — budgets are limits: absent = no truncation. All three limits are uniformly no-default — budgets are
host policy. `ResolvedExecutionLimits.maxOutputBytes` is `number | undefined`; host policy. `ResolvedExecutionLimits.maxOutputBytes` is `number | undefined`;
`boundOutput` only runs when the host set the limit. Explicit values validate as before `boundOutput` only runs when the host set the limit. Explicit values validate as before
(safe integer ≥ 0). (safe integer ≥ 0).
- OpenCode continues to pass NO limits, which now also means no CodeMode truncation. - OpenCode continues to pass NO limits, which now also means no CodeMode truncation.
`execute` is a normal `Tool.define` tool, so OpenCode's native tool-output truncation `execute` is a normal `Tool.define` tool, so OpenCode's native tool-output truncation
applies with no special-casing — verified by tracing `wrap()` (`tool.ts:130-144`, applies with no special-casing — verified by tracing `wrap()` (`tool.ts:130-144`,
50KB/2000-line thresholds in `truncate.ts`, full output dumped to a file under 50KB/2000-line thresholds in `truncate.ts`, full output dumped to a file under
`tool-output/`): the `metadata.truncated` self-truncation exemption never fires for `tool-output/`): the `metadata.truncated` self-truncation exemption never fires for
`execute` (its metadata never sets that key). One truncation layer, the host's — and it `execute` (its metadata never sets that key). One truncation layer, the host's — and it
is the richer one (file dump + explore/grep hint vs an inline marker). is the richer one (file dump + explore/grep hint vs an inline marker).
- Hosts without their own output bounding set `maxOutputBytes` explicitly; README table - Hosts without their own output bounding set `maxOutputBytes` explicitly; README table
and prose updated, adapter comment rewritten. Tests: codemode +1 (absent limit → 100KB and prose updated, adapter comment rewritten. Tests: codemode +1 (absent limit → 100KB
value + 50KB log line pass through unbounded, `truncated` undefined); the adapter test value + 50KB log line pass through unbounded, `truncated` undefined); the adapter test
that relied on the old default now asserts the oversized result reaches the shared that relied on the old default now asserts the oversized result reaches the shared
@@ -837,24 +856,25 @@ regular dependency; hosts depend on it themselves because the API surface is Eff
**Registry promotion + permission-aware catalog** (the "promote to a proper tool service" **Registry promotion + permission-aware catalog** (the "promote to a proper tool service"
restructure; fixes the §4 permission-advertising bug): restructure; fixes the §4 permission-advertising bug):
- **The adapter moved** `src/session/code-mode.ts` → `src/tool/code-mode.ts` and is now a
- **The adapter moved** `src/session/code-mode.ts` → `src/tool/code-mode.ts` and is now a
registry-resident tool service on the TaskTool precedent: `CodeModeTool = registry-resident tool service on the TaskTool precedent: `CodeModeTool =
Tool.define(CODE_MODE_TOOL, ...)` whose init depends on `MCP.Service`, `Agent.Service`, Tool.define(CODE_MODE_TOOL, ...)` whose init depends on `MCP.Service`, `Agent.Service`,
and `Session.Service`. It is yielded in `ToolRegistry.layer`, gated into `builtin` by and `Session.Service`. It is yielded in `ToolRegistry.layer`, gated into `builtin` by
`flags.experimentalCodeMode` (like the lsp/plan experiments), and `MCP.node` joined the `flags.experimentalCodeMode` (like the lsp/plan experiments), and `MCP.node` joined the
registry's `node.deps` (`MCP.node` has no ToolRegistry dependency, so no cycle). The registry's `node.deps` (`MCP.node` has no ToolRegistry dependency, so no cycle). The
session-level special-casing in `session/tools.ts` (ad-hoc `SessionCodeMode.define` + session-level special-casing in `session/tools.ts` (ad-hoc `SessionCodeMode.define` +
append) is deleted; the early return that suppresses raw per-MCP registration when the append) is deleted; the early return that suppresses raw per-MCP registration when the
flag is on stays session-side, keyed on the same flag+tool-count condition. flag is on stays session-side, keyed on the same flag+tool-count condition.
- **Enablement** lives in `ToolRegistry.tools()` next to the WebSearchTool check: the MCP - **Enablement** lives in `ToolRegistry.tools()` next to the WebSearchTool check: the MCP
tool count is consulted once (an Effect) before the synchronous filter, and code mode tool count is consulted once (an Effect) before the synchronous filter, and code mode
passes the predicate iff `flags.experimentalCodeMode` && count > 0. passes the predicate iff `flags.experimentalCodeMode` && count > 0.
- **Description split on the `describeTask` precedent**: the tool's static base - **Description split on the `describeTask` precedent**: the tool's static base
description is a two-line summary; `describeCodeMode(agent)` in `registry.tools()` description is a two-line summary; `describeCodeMode(agent)` in `registry.tools()`
appends the full CodeMode instructions (workflow/rules/syntax + grouped catalog, appends the full CodeMode instructions (workflow/rules/syntax + grouped catalog,
`catalogInstructions` in the adapter) at the same composition point as task — so `catalogInstructions` in the adapter) at the same composition point as task — so
`plugin.trigger("tool.definition")` sees the base description first. `plugin.trigger("tool.definition")` sees the base description first.
- **Permission-aware catalog + dispatch** (the bug fix): the visibility predicate from - **Permission-aware catalog + dispatch** (the bug fix): the visibility predicate from
`llm/request.ts` `resolveTools` is hoisted to `Permission.visibleTools(tools, ruleset)` `llm/request.ts` `resolveTools` is hoisted to `Permission.visibleTools(tools, ruleset)`
(a record filter over `Permission.disabled` — only a hard `deny` with pattern `"*"` (a record filter over `Permission.disabled` — only a hard `deny` with pattern `"*"`
hides a tool; ask-level rules stay fully visible and prompt at call time) and hides a tool; ask-level rules stay fully visible and prompt at call time) and
@@ -867,26 +887,27 @@ restructure; fixes the §4 permission-advertising bug):
even if the model guesses its name and yields the normal unknown-tool diagnostic. even if the model guesses its name and yields the normal unknown-tool diagnostic.
Documented gap (out of scope by design): per-message `user.tools[key] === false` arrives Documented gap (out of scope by design): per-message `user.tools[key] === false` arrives
at request-prep after descriptions are built and has no child-call equivalent. at request-prep after descriptions are built and has no child-call equivalent.
- **Preserved behavior**: cancellation race + pre-aborted-signal guard, `toSandboxResult` - **Preserved behavior**: cancellation race + pre-aborted-signal guard, `toSandboxResult`
unwrap order, attachment accumulation, `CODE_MODE_TOOL` at all title sites, no execution unwrap order, attachment accumulation, `CODE_MODE_TOOL` at all title sites, no execution
limits (native truncation only), `displayInput`, per-child `ctx.ask` gating (now wired limits (native truncation only), `displayInput`, per-child `ctx.ask` gating (now wired
through `Tool.Context` exactly like every registry tool). through `Tool.Context` exactly like every registry tool).
- **Explicit non-goal**: memoizing the catalog builder keyed on (ToolsChanged generation, - **Explicit non-goal**: memoizing the catalog builder keyed on (ToolsChanged generation,
permission ruleset) was considered and deliberately skipped — the per-turn rebuild is permission ruleset) was considered and deliberately skipped — the per-turn rebuild is
cheap (grouping + string rendering); revisit only if profiling shows it matters. cheap (grouping + string rendering); revisit only if profiling shows it matters.
- **Tests**: the two adapter suites moved to `test/tool/{code-mode,code-mode-integration} - **Tests**: the two adapter suites moved to `test/tool/{code-mode,code-mode-integration}
.test.ts` (mocked `MCP.Service`/`Agent.Service`/`Session.Service` replacing the direct .test.ts` (mocked `MCP.Service`/`Agent.Service`/`Session.Service` replacing the direct
`define(...)` construction; description assertions target `catalogInstructions`, the `define(...)` construction; description assertions target `catalogInstructions`, the
registry's composition input) and gained permission coverage: deny excluded from registry's composition input) and gained permission coverage: deny excluded from
catalog/search, ask-level stays visible and callable, denied tool undispatchable catalog/search, ask-level stays visible and callable, denied tool undispatchable
(unknown-tool diagnostic), `Permission.visibleTools` semantics. `test/tool/ (unknown-tool diagnostic), `Permission.visibleTools` semantics. `test/tool/
registry.test.ts` gained four registry-level tests: registered with flag+MCP tools, registry.test.ts` gained four registry-level tests: registered with flag+MCP tools,
excluded without MCP tools, excluded with flag off, and deny/ask catalog filtering excluded without MCP tools, excluded with flag off, and deny/ask catalog filtering
through `registry.tools()`. Suites: 43 + 16 adapter tests, 16 registry tests, all green. through `registry.tools()`. Suites: 43 + 16 adapter tests, 16 registry tests, all green.
**Shared MCP invocation middle (`McpInvoke.invoke`)** (closes the §4 "plugin hooks skip **Shared MCP invocation middle (`McpInvoke.invoke`)** (closes the §4 "plugin hooks skip
child calls" gap): child calls" gap):
- `packages/opencode/src/mcp/invoke.ts` extracts the duplicated "invoke an MCP tool"
- `packages/opencode/src/mcp/invoke.ts` extracts the duplicated "invoke an MCP tool"
middle into one shared `McpInvoke.invoke(input)`: plugin `tool.execute.before` hook → middle into one shared `McpInvoke.invoke(input)`: plugin `tool.execute.before` hook →
permission ask (`{ permission: key, patterns: ["*"], always: ["*"] }` via the caller's permission ask (`{ permission: key, patterns: ["*"], always: ["*"] }` via the caller's
`ctx.ask`) → dispatch through the ai-sdk tool's execute inside the `Tool.execute` `ctx.ask`) → dispatch through the ai-sdk tool's execute inside the `Tool.execute`
@@ -897,24 +918,24 @@ child calls" gap):
mode applies `toSandboxResult`. It lives under `src/mcp/` because both callers mode applies `toSandboxResult`. It lives under `src/mcp/` because both callers
already depend on MCP and the function is about invoking an MCP-backed ai-sdk tool, already depend on MCP and the function is about invoking an MCP-backed ai-sdk tool,
not about sessions or code mode. not about sessions or code mode.
- **After-hook payload**: fired inside `McpInvoke.invoke` with the raw MCP result — - **After-hook payload**: fired inside `McpInvoke.invoke` with the raw MCP result —
which is exactly what the legacy loop always passed (the raw `CallToolResult`, not which is exactly what the legacy loop always passed (the raw `CallToolResult`, not
the shaped `{title, output, metadata}`), so legacy behavior is preserved bit-for-bit the shaped `{title, output, metadata}`), so legacy behavior is preserved bit-for-bit
and the hook payload cannot drift between callers. No callback/edge-firing design and the hook payload cannot drift between callers. No callback/edge-firing design
was needed. was needed.
- **Synthetic child callID**: code-mode child calls pass `${parentCallID}/${n}` as the - **Synthetic child callID**: code-mode child calls pass `${parentCallID}/${n}` as the
hook/span callID (`parentCallID` = the `execute` call's `ctx.callID`, falling back to hook/span callID (`parentCallID` = the `execute` call's `ctx.callID`, falling back to
the entry key; `n` = per-execution counter starting at 1, shared across all child the entry key; `n` = per-execution counter starting at 1, shared across all child
calls in one program). callID is an opaque string — nothing parses it. The ai-sdk calls in one program). callID is an opaque string — nothing parses it. The ai-sdk
`toolCallId` (`options.toolCallId`) stays each caller's existing value `toolCallId` (`options.toolCallId`) stays each caller's existing value
(`ctx.callID ?? entry.key` for code mode). (`ctx.callID ?? entry.key` for code mode).
- **Child-scoped hook failures**: `CodeModeTool` (which now also yields - **Child-scoped hook failures**: `CodeModeTool` (which now also yields
`Plugin.Service`) wraps the whole child call — hooks, ask, dispatch — in `Plugin.Service`) wraps the whole child call — hooks, ask, dispatch — in
`toCatchable` (the generalization of the old `askPermission` catchCause), so a plugin `toCatchable` (the generalization of the old `askPermission` catchCause), so a plugin
hook failure fails ONLY that child call as a catchable in-program `toolError`; other hook failure fails ONLY that child call as a catchable in-program `toolError`; other
calls in the same program keep running and interruption still propagates as calls in the same program keep running and interruption still propagates as
interruption. Legacy semantics unchanged: a hook failure fails the tool call. interruption. Legacy semantics unchanged: a hook failure fails the tool call.
- **Tests**: `test/tool/code-mode.test.ts` +2 (child calls fire before/after with the - **Tests**: `test/tool/code-mode.test.ts` +2 (child calls fire before/after with the
MCP key and `parent/1`, `parent/2` ids, after hook carries the raw MCP result; a MCP key and `parent/1`, `parent/2` ids, after hook carries the raw MCP result; a
failing before hook is caught in-program, gates dispatch, and leaves the outer failing before hook is caught in-program, gates dispatch, and leaves the outer
execute ok) — both code-mode harnesses gained a `Plugin.Service` mock (pass-through execute ok) — both code-mode harnesses gained a `Plugin.Service` mock (pass-through
@@ -927,7 +948,8 @@ child calls" gap):
**Signature rendering + compound-assignment parity fixes** (externally reported, both **Signature rendering + compound-assignment parity fixes** (externally reported, both
verified real with failing tests before fixing): verified real with failing tests before fixing):
- **Non-identifier property names in rendered signatures** (`src/tool.ts`): `renderSchema`
- **Non-identifier property names in rendered signatures** (`src/tool.ts`): `renderSchema`
emitted raw property names, so schema properties like `foo-bar`/`@type`/`x.y`/`123` emitted raw property names, so schema properties like `foo-bar`/`@type`/`x.y`/`123`
rendered invalid TypeScript (`{ foo-bar?: string }`). Fixed with a `renderKey` helper — rendered invalid TypeScript (`{ foo-bar?: string }`). Fixed with a `renderKey` helper —
bare identifiers stay bare, everything else is `JSON.stringify`-quoted — applied in the bare identifiers stay bare, everything else is `JSON.stringify`-quoted — applied in the
@@ -936,13 +958,13 @@ verified real with failing tests before fixing):
bracket-notation `toolExpression` imports it: one source of truth for "is this a bare bracket-notation `toolExpression` imports it: one source of truth for "is this a bare
identifier" across object keys and tool paths. Tests: `signature.test.ts` +4 (compact, identifier" across object keys and tool paths. Tests: `signature.test.ts` +4 (compact,
pretty with JSDoc on a quoted key, JSON Schema input+output, Effect Schema struct). pretty with JSDoc on a quoted key, JSON Schema input+output, Effect Schema struct).
- **Numeric schema unions keep their real alternatives** (`src/tool.ts`): the old - **Numeric schema unions keep their real alternatives** (`src/tool.ts`): the old
`anyOf`/`oneOf` renderer collapsed any union containing `{ type: "number" }` to just `anyOf`/`oneOf` renderer collapsed any union containing `{ type: "number" }` to just
`number`, dropping real JSON Schema alternatives (`string | number`, `number | null`, `number`, dropping real JSON Schema alternatives (`string | number`, `number | null`,
etc.). The collapse is now restricted to Effect's number-schema artifact etc.). The collapse is now restricted to Effect's number-schema artifact
(`number | "NaN" | "Infinity" | "-Infinity"`, emitted as single-value string enums), (`number | "NaN" | "Infinity" | "-Infinity"`, emitted as single-value string enums),
while raw JSON Schema unions render every branch. Tests: `signature.test.ts` +3. while raw JSON Schema unions render every branch. Tests: `signature.test.ts` +3.
- **Compound assignment now matches binary-operator semantics** (`src/codemode.ts`): - **Compound assignment now matches binary-operator semantics** (`src/codemode.ts`):
`applyCompoundAssignment` did raw JS ops on interpreter wrapper objects, so `x += y` `applyCompoundAssignment` did raw JS ops on interpreter wrapper objects, so `x += y`
diverged from `x = x + y` (sandbox Date `d += 1` produced `"[object Object]1"`; diverged from `x = x + y` (sandbox Date `d += 1` produced `"[object Object]1"`;
`d -= 400` gave `NaN` instead of epoch arithmetic). The operator table + coercion moved `d -= 400` gave `NaN` instead of epoch arithmetic). The operator table + coercion moved
@@ -961,8 +983,10 @@ verified real with failing tests before fixing):
## 4. Remaining work (detailed TODO) ## 4. Remaining work (detailed TODO)
### Next DSL-expansion pass (done — see the DSL-expansion pass entry in §3) ### Next DSL-expansion pass (done — see the DSL-expansion pass entry in §3)
Batch these together — per user direction: important, but deliberately deferred to one Batch these together — per user direction: important, but deliberately deferred to one
focused interpreter-surface pass rather than picked off piecemeal. focused interpreter-surface pass rather than picked off piecemeal.
- [x] Medium-tier JS parity items deferred from the original audit: caught errors are plain - [x] Medium-tier JS parity items deferred from the original audit: caught errors are plain
`{ name, message }` objects, not `instanceof Error` (and `Error` isn't a value — `{ name, message }` objects, not `instanceof Error` (and `Error` isn't a value —
`x instanceof Error` is unsupported syntax); `splice` (still a `x instanceof Error` is unsupported syntax); `splice` (still a
@@ -981,6 +1005,7 @@ focused interpreter-surface pass rather than picked off piecemeal.
(`console.log({ m: map })`) — could deep-format instead. (`console.log({ m: map })`) — could deep-format instead.
### Next iteration: text-result handling (deliberate follow-up, user-directed) ### Next iteration: text-result handling (deliberate follow-up, user-directed)
- [ ] Revisit how MCP text results reach the program. Today: `structuredContent` when the - [ ] Revisit how MCP text results reach the program. Today: `structuredContent` when the
server sends it, else joined text as a plain string (the program JSON.parses it, server sends it, else joined text as a plain string (the program JSON.parses it,
guided by a workflow step). Considered and deferred: (a) conservative boundary guided by a workflow step). Considered and deferred: (a) conservative boundary
@@ -992,6 +1017,7 @@ focused interpreter-surface pass rather than picked off piecemeal.
revisit once real usage shows which failure modes matter. revisit once real usage shows which failure modes matter.
### Next iteration: stdlib surface (prioritized) ### Next iteration: stdlib surface (prioritized)
Current instructions say "usual Array/String/Object/Math/JSON methods," but the interpreter is Current instructions say "usual Array/String/Object/Math/JSON methods," but the interpreter is
intentionally a subset. Keep CodeMode focused on orchestration and data shaping, not a full host intentionally a subset. Keep CodeMode focused on orchestration and data shaping, not a full host
runtime, but close the high-friction gaps models are likely to reach for. runtime, but close the high-friction gaps models are likely to reach for.
@@ -1028,7 +1054,9 @@ Explicit non-goals for now: `structuredClone`, `WeakMap`/`WeakSet`, and timers
orchestration use case. orchestration use case.
### Wiring-review findings (subagent code review of the OpenCode integration, triaged) ### Wiring-review findings (subagent code review of the OpenCode integration, triaged)
Pre-PR fixes (user-approved cut): Pre-PR fixes (user-approved cut):
- [x] **Cancellation does not interrupt the interpreter** — the no-limits rationale claimed - [x] **Cancellation does not interrupt the interpreter** — the no-limits rationale claimed
"user cancel interrupts the execution fiber," but `tools.ts` runs tools via "user cancel interrupts the execution fiber," but `tools.ts` runs tools via
`run.promise` → `Effect.runPromise` (`effect/bridge.ts:64-66`) with NO abort wiring; `run.promise` → `Effect.runPromise` (`effect/bridge.ts:64-66`) with NO abort wiring;
@@ -1073,6 +1101,7 @@ Pre-PR fixes (user-approved cut):
`title: "execute"` sites in `code-mode.ts` now reference `CODE_MODE_TOOL`. `title: "execute"` sites in `code-mode.ts` now reference `CODE_MODE_TOOL`.
Post-MVP (logged, not blocking an experimental flag): Post-MVP (logged, not blocking an experimental flag):
- [x] **Plugin `tool.execute.before/after` hooks skip child calls** — legacy MCP - [x] **Plugin `tool.execute.before/after` hooks skip child calls** — legacy MCP
registration fires them per tool (`tools.ts:419-441`); under code mode only the registration fires them per tool (`tools.ts:419-441`); under code mode only the
outer `execute` fires them, so auditing/intercepting plugins silently lose MCP outer `execute` fires them, so auditing/intercepting plugins silently lose MCP
@@ -1105,6 +1134,7 @@ Post-MVP (logged, not blocking an experimental flag):
names that are no longer directly callable under code mode. names that are no longer directly callable under code mode.
### Backlog / loose ends (non-blocking, any order) ### Backlog / loose ends (non-blocking, any order)
- [ ] `evaluateUpdateExpression` (`++`/`--`) still uses raw `Number(current)`, so `d++` on a - [ ] `evaluateUpdateExpression` (`++`/`--`) still uses raw `Number(current)`, so `d++` on a
sandbox Date yields `NaN` where `d += 1` now uses epoch semantics (and real JS `d++` sandbox Date yields `NaN` where `d += 1` now uses epoch semantics (and real JS `d++`
would give epoch+0 numeric). Pre-existing, out of scope of the compound-assignment would give epoch+0 numeric). Pre-existing, out of scope of the compound-assignment
@@ -1141,7 +1171,7 @@ Post-MVP (logged, not blocking an experimental flag):
## 5. Context and gotchas for whoever picks this up ## 5. Context and gotchas for whoever picks this up
- **Motivating failure (why forgiving semantics + prompting matter):** in a real transcript, - **Motivating failure (why forgiving semantics + prompting matter):** in a real transcript,
the model wrote `me.result?.login ?? me.result` where the tool result was a JSON *string* the model wrote `me.result?.login ?? me.result` where the tool result was a JSON _string_
the old strict interpreter threw (`String property 'login' is not available`); then the the old strict interpreter threw (`String property 'login' is not available`); then the
model returned a raw 105KB payload, which native truncation dumped to a file, costing a model returned a raw 105KB payload, which native truncation dumped to a file, costing a
subagent round-trip to extract one number. Interpreter forgiveness stops the crashes; subagent round-trip to extract one number. Interpreter forgiveness stops the crashes;
File diff suppressed because it is too large Load Diff
+1 -7
View File
@@ -1,10 +1,4 @@
export { export { ToolError, CodeMode, ExecuteInputSchema, ExecuteResultSchema, toolError } from "./codemode.js"
ToolError,
CodeMode,
ExecuteInputSchema,
ExecuteResultSchema,
toolError,
} from "./codemode.js"
export { Tool } from "./tool.js" export { Tool } from "./tool.js"
export type { Definition as ToolDefinition, JsonSchema, ToolSchema } from "./tool.js" export type { Definition as ToolDefinition, JsonSchema, ToolSchema } from "./tool.js"
export type { ToolCallEnded, ToolCallHooks } from "./tool-runtime.js" export type { ToolCallEnded, ToolCallHooks } from "./tool-runtime.js"
+134 -51
View File
@@ -21,10 +21,15 @@ export type HostTools<R = never> = {
export type Services<Tools> = Tools extends (...args: Array<unknown>) => Effect.Effect<unknown, unknown, infer R> export type Services<Tools> = Tools extends (...args: Array<unknown>) => Effect.Effect<unknown, unknown, infer R>
? R ? R
: Tools extends { readonly _tag: "CodeModeTool"; readonly run: (input: unknown) => Effect.Effect<unknown, unknown, infer R> } : Tools extends {
readonly _tag: "CodeModeTool"
readonly run: (input: unknown) => Effect.Effect<unknown, unknown, infer R>
}
? R ? R
: Tools extends object : Tools extends object
? string extends keyof Tools ? never : Services<Tools[keyof Tools]> ? string extends keyof Tools
? never
: Services<Tools[keyof Tools]>
: never : never
/** Minimal audit record retained for each admitted tool call. */ /** Minimal audit record retained for each admitted tool call. */
@@ -68,11 +73,13 @@ export type SafeObject = Record<string, unknown>
const reservedNamespace = "$codemode" const reservedNamespace = "$codemode"
const defaultMaxInlineCatalogTokens = 2_000 const defaultMaxInlineCatalogTokens = 2_000
const defaultSearchLimit = 10 const defaultSearchLimit = 10
const searchSignature = "tools.$codemode.search({ query?: string, namespace?: string, limit?: number }): Promise<{ items: Array<{ path: string; description: string; signature: string }>; total: number }>" const searchSignature =
"tools.$codemode.search({ query?: string, namespace?: string, limit?: number }): Promise<{ items: Array<{ path: string; description: string; signature: string }>; total: number }>"
const toolExpression = (path: string) => const toolExpression = (path: string) =>
"tools" + path "tools" +
path
.split(".") .split(".")
.map((segment) => identifierSegment.test(segment) ? `.${segment}` : `[${JSON.stringify(segment)}]`) .map((segment) => (identifierSegment.test(segment) ? `.${segment}` : `[${JSON.stringify(segment)}]`))
.join("") .join("")
export class ToolReference { export class ToolReference {
@@ -88,7 +95,12 @@ const MAX_VALUE_DEPTH = 32
export class ToolRuntimeError extends Error { export class ToolRuntimeError extends Error {
constructor( constructor(
readonly kind: "UnknownTool" | "InvalidToolInput" | "InvalidToolOutput" | "InvalidDataValue" | "ToolCallLimitExceeded", readonly kind:
| "UnknownTool"
| "InvalidToolInput"
| "InvalidToolOutput"
| "InvalidDataValue"
| "ToolCallLimitExceeded",
message: string, message: string,
readonly suggestions: ReadonlyArray<string> = [], readonly suggestions: ReadonlyArray<string> = [],
) { ) {
@@ -131,7 +143,13 @@ export const isBlockedMember = (name: string): boolean => blockedMemberNames.has
export const copyIn = (value: unknown, label: string, preserveSandboxValues = false): unknown => export const copyIn = (value: unknown, label: string, preserveSandboxValues = false): unknown =>
copyBounded(value, label, 0, new Set(), preserveSandboxValues) copyBounded(value, label, 0, new Set(), preserveSandboxValues)
const copyBounded = (value: unknown, label: string, depth: number, seen: Set<object>, preserveSandboxValues: boolean): unknown => { const copyBounded = (
value: unknown,
label: string,
depth: number,
seen: Set<object>,
preserveSandboxValues: boolean,
): unknown => {
if (depth > MAX_VALUE_DEPTH) { if (depth > MAX_VALUE_DEPTH) {
throw new ToolRuntimeError("InvalidDataValue", `${label} exceeds the maximum value depth of ${MAX_VALUE_DEPTH}.`) throw new ToolRuntimeError("InvalidDataValue", `${label} exceeds the maximum value depth of ${MAX_VALUE_DEPTH}.`)
} }
@@ -166,7 +184,12 @@ const copyBounded = (value: unknown, label: string, depth: number, seen: Set<obj
// Intra-sandbox checkpoints keep sandbox value instances alive as leaves; their contents // Intra-sandbox checkpoints keep sandbox value instances alive as leaves; their contents
// are never walked here (Map/Set members are validated where mutation happens, and the // are never walked here (Map/Set members are validated where mutation happens, and the
// real boundary still serializes them below). // real boundary still serializes them below).
if (value instanceof SandboxDate || value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet) { if (
value instanceof SandboxDate ||
value instanceof SandboxRegExp ||
value instanceof SandboxMap ||
value instanceof SandboxSet
) {
return value return value
} }
// Host instances cannot normally reach an intra-sandbox checkpoint (tool results cross // Host instances cannot normally reach an intra-sandbox checkpoint (tool results cross
@@ -197,8 +220,12 @@ const copyBounded = (value: unknown, label: string, depth: number, seen: Set<obj
return Number.isFinite(value.getTime()) ? value.toISOString() : null return Number.isFinite(value.getTime()) ? value.toISOString() : null
} }
if ( if (
value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet || value instanceof SandboxRegExp ||
value instanceof RegExp || value instanceof Map || value instanceof Set value instanceof SandboxMap ||
value instanceof SandboxSet ||
value instanceof RegExp ||
value instanceof Map ||
value instanceof Set
) { ) {
return Object.create(null) as SafeObject return Object.create(null) as SafeObject
} }
@@ -250,7 +277,10 @@ export const copyOut = (value: unknown, undefinedAsNull = false): unknown => {
return value return value
} }
const definitions = <R>(tools: HostTools<R>, path: ReadonlyArray<string> = []): Array<{ path: string; definition: Definition<R> }> => { const definitions = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string> = [],
): Array<{ path: string; definition: Definition<R> }> => {
const entries: Array<{ path: string; definition: Definition<R> }> = [] const entries: Array<{ path: string; definition: Definition<R> }> = []
for (const [name, value] of Object.entries(tools)) { for (const [name, value] of Object.entries(tools)) {
const next = [...path, name] const next = [...path, name]
@@ -342,8 +372,11 @@ const toSearchEntry = <R>(path: string, definition: Definition<R>, description:
path, path,
definition.description, definition.description,
...inputProperties(definition).flatMap(({ name, description: property }) => ...inputProperties(definition).flatMap(({ name, description: property }) =>
property === undefined ? [name] : [name, property]), property === undefined ? [name] : [name, property],
].join("\n").toLowerCase(), ),
]
.join("\n")
.toLowerCase(),
}) })
/** The runtime search index over every described tool. Search is always registered. */ /** The runtime search index over every described tool. Search is always registered. */
@@ -396,7 +429,8 @@ export const discoveryPlan = <R>(
namespace, namespace,
picked: new Set<ToolDescription>(), picked: new Set<ToolDescription>(),
queue: [...group].sort( queue: [...group].sort(
(left, right) => estimate(catalogLine(left)) - estimate(catalogLine(right)) || left.path.localeCompare(right.path), (left, right) =>
estimate(catalogLine(left)) - estimate(catalogLine(right)) || left.path.localeCompare(right.path),
), ),
})) }))
let used = 0 let used = 0
@@ -472,7 +506,9 @@ export const discoveryPlan = <R>(
"- A result typed `Promise<unknown>` has no guaranteed shape — verify what actually came back before relying on its fields.", "- A result typed `Promise<unknown>` has no guaranteed shape — verify what actually came back before relying on its fields.",
"- Run independent calls in parallel: `await Promise.all(items.map((item) => tools.<namespace>.<tool>(item)))`.", "- Run independent calls in parallel: `await Promise.all(items.map((item) => tools.<namespace>.<tool>(item)))`.",
"- `Object.keys(tools)` lists namespaces; `Object.keys(tools.<namespace>)` lists its tools; `for...in` works on both.", "- `Object.keys(tools)` lists namespaces; `Object.keys(tools.<namespace>)` lists its tools; `for...in` works on both.",
...(complete ? [] : ['- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.']), ...(complete
? []
: ['- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.']),
] ]
const syntax = [ const syntax = [
@@ -500,26 +536,21 @@ export const discoveryPlan = <R>(
const count = `${group.length} tool${group.length === 1 ? "" : "s"}` const count = `${group.length} tool${group.length === 1 ? "" : "s"}`
// Annotate only when a namespace is not fully shown, so a comprehensive // Annotate only when a namespace is not fully shown, so a comprehensive
// namespace reads cleanly and a truncated one is unambiguous. // namespace reads cleanly and a truncated one is unambiguous.
const label = picked.size === group.length ? count : picked.size === 0 ? `${count}, none shown` : `${count}, ${picked.size} shown` const label =
picked.size === group.length
? count
: picked.size === 0
? `${count}, none shown`
: `${count}, ${picked.size} shown`
toolSection.push(`- ${namespace} (${label})`) toolSection.push(`- ${namespace} (${label})`)
for (const tool of group) if (picked.has(tool)) toolSection.push(catalogLine(tool)) for (const tool of group) if (picked.has(tool)) toolSection.push(catalogLine(tool))
} }
if (!complete) { if (!complete) {
toolSection.push( toolSection.push("", "Search returns complete callable signatures:", `- ${searchSignature}`)
"",
"Search returns complete callable signatures:",
`- ${searchSignature}`,
)
} }
} }
const lines = [ const lines = [...intro, ...workflow, ...rules, ...syntax, ...toolSection]
...intro,
...workflow,
...rules,
...syntax,
...toolSection,
]
return { return {
catalog: described, catalog: described,
instructions: lines.join("\n"), instructions: lines.join("\n"),
@@ -534,18 +565,29 @@ export const discoveryPlan = <R>(
* function in JS). An unknown path is an `UnknownTool` error pointing at the working * function in JS). An unknown path is an `UnknownTool` error pointing at the working
* discovery idioms, mirroring how calling an unknown tool fails. * discovery idioms, mirroring how calling an unknown tool fails.
*/ */
const namespaceKeys = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, searchEnabled: boolean): ReadonlyArray<string> => { const namespaceKeys = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string>,
searchEnabled: boolean,
): ReadonlyArray<string> => {
// The reserved discovery namespace is virtual (never present in the host tree); enumerate // The reserved discovery namespace is virtual (never present in the host tree); enumerate
// it explicitly so `Object.keys(tools.$codemode)` matches the callable surface. // it explicitly so `Object.keys(tools.$codemode)` matches the callable surface.
if (searchEnabled && path.length === 1 && path[0] === reservedNamespace) return ["search"] if (searchEnabled && path.length === 1 && path[0] === reservedNamespace) return ["search"]
let value: HostTool<R> | Definition<R> | HostTools<R> = tools let value: HostTool<R> | Definition<R> | HostTools<R> = tools
for (const segment of path) { for (const segment of path) {
if (isBlockedMember(segment) || typeof value === "function" || isDefinition(value) || !Object.hasOwn(value, segment)) { if (
isBlockedMember(segment) ||
typeof value === "function" ||
isDefinition(value) ||
!Object.hasOwn(value, segment)
) {
throw new ToolRuntimeError( throw new ToolRuntimeError(
"UnknownTool", "UnknownTool",
`Unknown tool namespace '${path.join(".")}'.`, `Unknown tool namespace '${path.join(".")}'.`,
searchEnabled searchEnabled
? ["Object.keys(tools) lists the available namespaces; tools.$codemode.search({ query }) finds described tools."] ? [
"Object.keys(tools) lists the available namespaces; tools.$codemode.search({ query }) finds described tools.",
]
: ["Object.keys(tools) lists the available namespaces."], : ["Object.keys(tools) lists the available namespaces."],
) )
} }
@@ -555,12 +597,25 @@ const namespaceKeys = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, sear
return Object.keys(value) return Object.keys(value)
} }
const resolve = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, searchEnabled: boolean): HostTool<R> | Definition<R> => { const resolve = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string>,
searchEnabled: boolean,
): HostTool<R> | Definition<R> => {
let value: HostTool<R> | Definition<R> | HostTools<R> = tools let value: HostTool<R> | Definition<R> | HostTools<R> = tools
for (const segment of path) { for (const segment of path) {
if (isBlockedMember(segment) || typeof value === "function" || isDefinition(value) || !Object.hasOwn(value, segment)) { if (
throw new ToolRuntimeError("UnknownTool", `Unknown tool '${path.join(".")}'.`, searchEnabled ? ["Use tools.$codemode.search({ query }) to find available described tools."] : []) isBlockedMember(segment) ||
typeof value === "function" ||
isDefinition(value) ||
!Object.hasOwn(value, segment)
) {
throw new ToolRuntimeError(
"UnknownTool",
`Unknown tool '${path.join(".")}'.`,
searchEnabled ? ["Use tools.$codemode.search({ query }) to find available described tools."] : [],
)
} }
value = value[segment] as HostTool<R> | Definition<R> | HostTools<R> value = value[segment] as HostTool<R> | Definition<R> | HostTools<R>
} }
@@ -602,7 +657,8 @@ export const make = <R>(
return effect.pipe( return effect.pipe(
Effect.tap(() => onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "success" })), Effect.tap(() => onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "success" })),
Effect.tapError((error) => Effect.tapError((error) =>
onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "failure", message: failureMessage(error) })), onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "failure", message: failureMessage(error) }),
),
) )
} }
@@ -624,7 +680,7 @@ export const make = <R>(
calls, calls,
keys: (path) => namespaceKeys(tools, path, searchEnabled), keys: (path) => namespaceKeys(tools, path, searchEnabled),
invoke: (path, args) => invoke: (path, args) =>
Effect.gen(function*() { Effect.gen(function* () {
const name = path.join(".") const name = path.join(".")
const externalArgs = args.map((arg) => copyOut(copyIn(arg, `Arguments for tool '${name}'`))) const externalArgs = args.map((arg) => copyOut(copyIn(arg, `Arguments for tool '${name}'`)))
const call = { name } const call = { name }
@@ -637,17 +693,32 @@ export const make = <R>(
if (!searchEnabled) throw new ToolRuntimeError("UnknownTool", `Unknown tool '${name}'.`) if (!searchEnabled) throw new ToolRuntimeError("UnknownTool", `Unknown tool '${name}'.`)
const input = externalArgs[0] const input = externalArgs[0]
if (externalArgs.length !== 1 || input === null || typeof input !== "object" || Array.isArray(input)) { if (externalArgs.length !== 1 || input === null || typeof input !== "object" || Array.isArray(input)) {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search expects { query?: string; namespace?: string; limit?: number }.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search expects { query?: string; namespace?: string; limit?: number }.",
)
} }
const request = input as { query?: unknown; namespace?: unknown; limit?: unknown } const request = input as { query?: unknown; namespace?: unknown; limit?: unknown }
if (request.query !== undefined && typeof request.query !== "string") { if (request.query !== undefined && typeof request.query !== "string") {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search query must be a string when provided.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search query must be a string when provided.",
)
} }
if (request.namespace !== undefined && typeof request.namespace !== "string") { if (request.namespace !== undefined && typeof request.namespace !== "string") {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search namespace must be a string when provided.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search namespace must be a string when provided.",
)
} }
if (request.limit !== undefined && (typeof request.limit !== "number" || !Number.isSafeInteger(request.limit) || request.limit <= 0)) { if (
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search limit must be a positive safe integer when provided.") request.limit !== undefined &&
(typeof request.limit !== "number" || !Number.isSafeInteger(request.limit) || request.limit <= 0)
) {
throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search limit must be a positive safe integer when provided.",
)
} }
const query = typeof request.query === "string" ? request.query : "" const query = typeof request.query === "string" ? request.query : ""
const namespace = typeof request.namespace === "string" ? request.namespace : undefined const namespace = typeof request.namespace === "string" ? request.namespace : undefined
@@ -656,20 +727,27 @@ export const make = <R>(
Effect.try({ Effect.try({
try: () => { try: () => {
const limit = typeof request.limit === "number" ? request.limit : defaultSearchLimit const limit = typeof request.limit === "number" ? request.limit : defaultSearchLimit
const scoped = namespace === undefined ? searchIndex : searchIndex.filter((entry) => entry.namespace === namespace) const scoped =
namespace === undefined ? searchIndex : searchIndex.filter((entry) => entry.namespace === namespace)
// A query that names one tool path exactly (canonical path or rendered // A query that names one tool path exactly (canonical path or rendered
// JavaScript expression) is a lookup, not a search: return that tool alone. // JavaScript expression) is a lookup, not a search: return that tool alone.
const trimmed = query.trim() const trimmed = query.trim()
const pathQuery = trimmed.startsWith("tools.") ? trimmed.slice("tools.".length) : trimmed const pathQuery = trimmed.startsWith("tools.") ? trimmed.slice("tools.".length) : trimmed
const exact = pathQuery === "" ? undefined : scoped.find((entry) => const exact =
entry.description.path === pathQuery || toolExpression(entry.description.path) === trimmed) pathQuery === ""
? undefined
: scoped.find(
(entry) =>
entry.description.path === pathQuery || toolExpression(entry.description.path) === trimmed,
)
const terms = tokenize(query).map(termForms) const terms = tokenize(query).map(termForms)
// Additive field-weighted scoring, summed across terms: exact path or path // Additive field-weighted scoring, summed across terms: exact path or path
// segment (20) > path substring (8) > description substring (4) > any // segment (20) > path substring (8) > description substring (4) > any
// searchable text, incl. input parameter names/descriptions (2). Each term // searchable text, incl. input parameter names/descriptions (2). Each term
// matches a field when any of its forms (the term or a singular variant) // matches a field when any of its forms (the term or a singular variant)
// does. An empty query browses everything, alphabetical by path. // does. An empty query browses everything, alphabetical by path.
const ranked = exact !== undefined const ranked =
exact !== undefined
? [exact] ? [exact]
: scoped : scoped
.map((entry) => { .map((entry) => {
@@ -687,8 +765,11 @@ export const make = <R>(
return { entry, score } return { entry, score }
}) })
.filter(({ score }) => terms.length === 0 || score > 0) .filter(({ score }) => terms.length === 0 || score > 0)
.sort((left, right) => .sort(
right.score - left.score || left.entry.description.path.localeCompare(right.entry.description.path)) (left, right) =>
right.score - left.score ||
left.entry.description.path.localeCompare(right.entry.description.path),
)
.map(({ entry }) => entry) .map(({ entry }) => entry)
// Result paths are rendered as JavaScript expressions so each `path` is // Result paths are rendered as JavaScript expressions so each `path` is
// directly usable as the call site (`await tools.github.list({ ... })` or // directly usable as the call site (`await tools.github.list({ ... })` or
@@ -711,10 +792,12 @@ export const make = <R>(
const tool = resolve(tools, path, searchEnabled) const tool = resolve(tools, path, searchEnabled)
let describedInput: unknown let describedInput: unknown
if (isDefinition(tool)) { if (isDefinition(tool)) {
if (externalArgs.length !== 1) throw new ToolRuntimeError("InvalidToolInput", `Tool '${name}' expects exactly one input object.`) if (externalArgs.length !== 1)
throw new ToolRuntimeError("InvalidToolInput", `Tool '${name}' expects exactly one input object.`)
describedInput = yield* Effect.try({ describedInput = yield* Effect.try({
try: () => decodeToolInput(tool, externalArgs[0]), try: () => decodeToolInput(tool, externalArgs[0]),
catch: (cause) => new ToolRuntimeError("InvalidToolInput", `Invalid input for tool '${name}': ${String(cause)}`), catch: (cause) =>
new ToolRuntimeError("InvalidToolInput", `Invalid input for tool '${name}': ${String(cause)}`),
}) })
} }
const input = isDefinition(tool) ? describedInput : externalArgs const input = isDefinition(tool) ? describedInput : externalArgs
@@ -722,7 +805,7 @@ export const make = <R>(
const currentCall = { index, name, input } const currentCall = { index, name, input }
if (isDefinition(tool)) { if (isDefinition(tool)) {
return yield* observeEnd( return yield* observeEnd(
Effect.gen(function*() { Effect.gen(function* () {
const raw = yield* runHost(Effect.suspend(() => tool.run(describedInput))) const raw = yield* runHost(Effect.suspend(() => tool.run(describedInput)))
const result = yield* Effect.try({ const result = yield* Effect.try({
try: () => decodeToolOutput(tool, raw), try: () => decodeToolOutput(tool, raw),
@@ -734,7 +817,7 @@ export const make = <R>(
) )
} }
return yield* observeEnd( return yield* observeEnd(
Effect.gen(function*() { Effect.gen(function* () {
return yield* decodeOutput(yield* runHost(Effect.suspend(() => tool(...externalArgs))), name) return yield* decodeOutput(yield* runHost(Effect.suspend(() => tool(...externalArgs))), name)
}), }),
currentCall, currentCall,
+27 -12
View File
@@ -57,8 +57,7 @@ export type Options<I extends ToolSchema, O extends ToolSchema | undefined, R =
export const isDefinition = <R = never>(value: unknown): value is Definition<R> => export const isDefinition = <R = never>(value: unknown): value is Definition<R> =>
typeof value === "object" && value !== null && "_tag" in value && value._tag === "CodeModeTool" typeof value === "object" && value !== null && "_tag" in value && value._tag === "CodeModeTool"
const isEffectSchema = (schema: ToolSchema): schema is Schema.Decoder<unknown> & Schema.Top => const isEffectSchema = (schema: ToolSchema): schema is Schema.Decoder<unknown> & Schema.Top => Schema.isSchema(schema)
Schema.isSchema(schema)
const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unknown" const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unknown"
@@ -69,10 +68,12 @@ const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unkn
export const identifierSegment = /^[A-Za-z_$][A-Za-z0-9_$]*$/ export const identifierSegment = /^[A-Za-z_$][A-Za-z0-9_$]*$/
/** Renders a property name as a valid TS object key: bare when an identifier, quoted otherwise. */ /** Renders a property name as a valid TS object key: bare when an identifier, quoted otherwise. */
const renderKey = (name: string): string => identifierSegment.test(name) ? name : JSON.stringify(name) const renderKey = (name: string): string => (identifierSegment.test(name) ? name : JSON.stringify(name))
const effectNumberSentinel = (schema: JsonSchema) => const effectNumberSentinel = (schema: JsonSchema) =>
schema.type === "string" && Array.isArray(schema.enum) && schema.enum.length === 1 && schema.type === "string" &&
Array.isArray(schema.enum) &&
schema.enum.length === 1 &&
(schema.enum[0] === "NaN" || schema.enum[0] === "Infinity" || schema.enum[0] === "-Infinity") (schema.enum[0] === "NaN" || schema.enum[0] === "Infinity" || schema.enum[0] === "-Infinity")
/** /**
@@ -118,7 +119,8 @@ const docTags = (schema: JsonSchema): Array<string> => {
*/ */
const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad: string): string => { const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad: string): string => {
const lines = [...(description === undefined ? [] : description.split("\n")), ...tags].map((line) => const lines = [...(description === undefined ? [] : description.split("\n")), ...tags].map((line) =>
line.replaceAll("*/", "* /").replace(/\s+$/, "")) line.replaceAll("*/", "* /").replace(/\s+$/, ""),
)
while (lines.length > 0 && lines[0]!.trim() === "") lines.shift() while (lines.length > 0 && lines[0]!.trim() === "") lines.shift()
while (lines.length > 0 && lines[lines.length - 1]!.trim() === "") lines.pop() while (lines.length > 0 && lines[lines.length - 1]!.trim() === "") lines.pop()
if (lines.length === 0) return "" if (lines.length === 0) return ""
@@ -127,7 +129,12 @@ const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad
return `${pad}/**\n${body}\n${pad} */\n` return `${pad}/**\n${body}\n${pad} */\n`
} }
const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: ReadonlySet<string> = new Set()): string => { const renderSchema = (
schema: JsonSchema,
ctx: RenderContext,
depth = 0,
seen: ReadonlySet<string> = new Set(),
): string => {
if (depth > MAX_RENDER_DEPTH) return "unknown" if (depth > MAX_RENDER_DEPTH) return "unknown"
if (schema.$ref) { if (schema.$ref) {
const name = schema.$ref.split("/").pop() const name = schema.$ref.split("/").pop()
@@ -146,13 +153,16 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
if ( if (
alternatives.some((item) => item.type === "number") && alternatives.some((item) => item.type === "number") &&
alternatives.every((item) => item.type === "number" || effectNumberSentinel(item)) alternatives.every((item) => item.type === "number" || effectNumberSentinel(item))
) return "number" )
return "number"
// An empty Schema.Struct({}) emits `anyOf: [{ type: "object" }, { type: "array" }]` // An empty Schema.Struct({}) emits `anyOf: [{ type: "object" }, { type: "array" }]`
// (no properties/items); render the bare shape as {} instead of `{} | Array<unknown>`. // (no properties/items); render the bare shape as {} instead of `{} | Array<unknown>`.
if ( if (
alternatives.length === 2 && alternatives.length === 2 &&
alternatives[0]?.type === "object" && alternatives[0].properties === undefined && alternatives[0]?.type === "object" &&
alternatives[1]?.type === "array" && alternatives[1].items === undefined alternatives[0].properties === undefined &&
alternatives[1]?.type === "array" &&
alternatives[1].items === undefined
) { ) {
return "{}" return "{}"
} }
@@ -170,7 +180,8 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
const required = new Set(schema.required ?? []) const required = new Set(schema.required ?? [])
const properties = Object.entries(schema.properties ?? {}) const properties = Object.entries(schema.properties ?? {})
const additional = schema.additionalProperties const additional = schema.additionalProperties
const indexType = additional && typeof additional === "object" ? renderSchema(additional, ctx, depth + 1, seen) : undefined const indexType =
additional && typeof additional === "object" ? renderSchema(additional, ctx, depth + 1, seen) : undefined
const field = ([name, value]: readonly [string, JsonSchema]) => const field = ([name, value]: readonly [string, JsonSchema]) =>
`${renderKey(name)}${required.has(name) ? "" : "?"}: ${renderSchema(value, ctx, depth + 1, seen)}` `${renderKey(name)}${required.has(name) ? "" : "?"}: ${renderSchema(value, ctx, depth + 1, seen)}`
@@ -183,7 +194,9 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
// Pretty: an indented block, each described field preceded by its JSDoc comment. // Pretty: an indented block, each described field preceded by its JSDoc comment.
if (properties.length === 0 && indexType === undefined) return "{}" if (properties.length === 0 && indexType === undefined) return "{}"
const pad = " ".repeat(depth + 1) const pad = " ".repeat(depth + 1)
const lines = properties.map((entry) => `${jsdoc(entry[1].description, docTags(entry[1]), pad)}${pad}${field(entry)}`) const lines = properties.map(
(entry) => `${jsdoc(entry[1].description, docTags(entry[1]), pad)}${pad}${field(entry)}`,
)
if (indexType !== undefined) lines.push(`${pad}[key: string]: ${indexType}`) if (indexType !== undefined) lines.push(`${pad}[key: string]: ${indexType}`)
return `{\n${lines.join("\n")}\n${" ".repeat(depth)}}` return `{\n${lines.join("\n")}\n${" ".repeat(depth)}}`
} }
@@ -262,7 +275,9 @@ export const inputProperties = <R>(definition: Definition<R>): Array<InputProper
* fields; the default stays the compact single-line form. * fields; the default stays the compact single-line form.
*/ */
export const inputTypeScript = <R>(definition: Definition<R>, pretty = false): string => export const inputTypeScript = <R>(definition: Definition<R>, pretty = false): string =>
isEffectSchema(definition.input) ? toTypeScript(definition.input, false, pretty) : jsonSchemaToTypeScript(definition.input, pretty) isEffectSchema(definition.input)
? toTypeScript(definition.input, false, pretty)
: jsonSchemaToTypeScript(definition.input, pretty)
/** /**
* The model-visible TypeScript type of a tool's result; tools without an output schema * The model-visible TypeScript type of a tool's result; tools without an output schema
+4 -1
View File
@@ -49,4 +49,7 @@ export class SandboxSet {
} }
export const isSandboxValue = (value: unknown): value is SandboxDate | SandboxRegExp | SandboxMap | SandboxSet => export const isSandboxValue = (value: unknown): value is SandboxDate | SandboxRegExp | SandboxMap | SandboxSet =>
value instanceof SandboxDate || value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet value instanceof SandboxDate ||
value instanceof SandboxRegExp ||
value instanceof SandboxMap ||
value instanceof SandboxSet
+209 -114
View File
@@ -1,6 +1,13 @@
import { describe, expect, test } from "bun:test" import { describe, expect, test } from "bun:test"
import { Cause, Effect, Schema } from "effect" import { Cause, Effect, Schema } from "effect"
import { CodeMode, ExecuteInputSchema, ExecuteResultSchema, Tool, toolError, type ExecutionLimits } from "../src/index.js" import {
CodeMode,
ExecuteInputSchema,
ExecuteResultSchema,
Tool,
toolError,
type ExecutionLimits,
} from "../src/index.js"
import type { Definition } from "../src/tool.js" import type { Definition } from "../src/tool.js"
const run = (tool: Definition<never>) => const run = (tool: Definition<never>) =>
@@ -75,11 +82,14 @@ describe("CodeMode host failure boundary", () => {
output: Schema.Unknown, output: Schema.Unknown,
run: () => run: () =>
Effect.succeed( Effect.succeed(
new Proxy({}, { new Proxy(
{},
{
ownKeys: () => { ownKeys: () => {
throw new Error("host-output-secret") throw new Error("host-output-secret")
}, },
}), },
),
), ),
}), }),
) )
@@ -162,9 +172,7 @@ describe("CodeMode tool-call observation", () => {
) )
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(calls).toStrictEqual([ expect(calls).toStrictEqual([{ index: 0, name: "context.lookup", input: { query: "deployment failure" } }])
{ index: 0, name: "context.lookup", input: { query: "deployment failure" } },
])
}) })
test("observes settled calls with outcome and duration", async () => { test("observes settled calls with outcome and duration", async () => {
@@ -173,16 +181,17 @@ describe("CodeMode tool-call observation", () => {
description: "Look up a value", description: "Look up a value",
input: Schema.Struct({ query: Schema.String }), input: Schema.Struct({ query: Schema.String }),
output: Schema.String, output: Schema.String,
run: ({ query }) => run: ({ query }) => (query === "boom" ? Effect.fail(toolError("Lookup refused")) : Effect.succeed(query)),
query === "boom" ? Effect.fail(toolError("Lookup refused")) : Effect.succeed(query),
}) })
const runtime = CodeMode.make({ const runtime = CodeMode.make({
tools: { context: { lookup } }, tools: { context: { lookup } },
onToolCallStart: (call) => Effect.sync(() => { onToolCallStart: (call) =>
Effect.sync(() => {
events.push({ phase: "start", index: call.index, name: call.name }) events.push({ phase: "start", index: call.index, name: call.name })
}), }),
onToolCallEnd: (call) => Effect.sync(() => { onToolCallEnd: (call) =>
Effect.sync(() => {
expect(call.durationMs).toBeGreaterThanOrEqual(0) expect(call.durationMs).toBeGreaterThanOrEqual(0)
events.push({ events.push({
phase: "end", phase: "end",
@@ -210,13 +219,15 @@ describe("CodeMode tool-call observation", () => {
describe("CodeMode console capture", () => { describe("CodeMode console capture", () => {
test("captures console output as bounded result logs", async () => { test("captures console output as bounded result logs", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
const returned = console.log("Thread info:", { name: "Demo", count: 2 }) const returned = console.log("Thread info:", { name: "Demo", count: 2 })
console.warn("careful") console.warn("careful")
return returned return returned
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
@@ -228,39 +239,45 @@ describe("CodeMode console capture", () => {
}) })
test("keeps logs captured before failures", async () => { test("keeps logs captured before failures", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log("before failure") console.log("before failure")
throw new Error("boom") throw new Error("boom")
`, `,
})) }),
)
expect(result.ok ? undefined : result.logs).toStrictEqual(["before failure"]) expect(result.ok ? undefined : result.logs).toStrictEqual(["before failure"])
expect(result.ok ? undefined : result.error.message).toBe("Uncaught: boom") expect(result.ok ? undefined : result.error.message).toBe("Uncaught: boom")
}) })
test("prints NaN and Infinity literally instead of the JSON null", async () => { test("prints NaN and Infinity literally instead of the JSON null", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log(NaN) console.log(NaN)
console.log(Infinity, -Infinity) console.log(Infinity, -Infinity)
console.log({ ratio: NaN, bounds: [Infinity] }) console.log({ ratio: NaN, bounds: [Infinity] })
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual(["NaN", "Infinity -Infinity", '{"ratio":NaN,"bounds":[Infinity]}']) expect(result.logs).toStrictEqual(["NaN", "Infinity -Infinity", '{"ratio":NaN,"bounds":[Infinity]}'])
}) })
test("renders sandbox values nested inside logged containers", async () => { test("renders sandbox values nested inside logged containers", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log({ m: new Map([["a", 1]]), when: new Date(0), r: /ab/g, s: new Set([1, 2]) }) console.log({ m: new Map([["a", 1]]), when: new Date(0), r: /ab/g, s: new Set([1, 2]) })
console.log([new Date(0)]) console.log([new Date(0)])
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual([ expect(result.logs).toStrictEqual([
@@ -270,7 +287,8 @@ describe("CodeMode console capture", () => {
}) })
test("console formatting is total: cycles and opaque references render as markers", async () => { test("console formatting is total: cycles and opaque references render as markers", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
const m = new Map() const m = new Map()
m.set("self", m) m.set("self", m)
@@ -278,29 +296,30 @@ describe("CodeMode console capture", () => {
console.log({ fn: (x) => x, ok: 1 }) console.log({ fn: (x) => x, ok: 1 })
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual([ expect(result.logs).toStrictEqual(['{"box":Map(1) [["self",[Circular]]]}', '{"fn":[CodeMode reference],"ok":1}'])
'{"box":Map(1) [["self",[Circular]]]}',
'{"fn":[CodeMode reference],"ok":1}',
])
}) })
test("console.table renders sandbox value cells", async () => { test("console.table renders sandbox value cells", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.table([{ when: new Date(0), n: NaN }]) console.table([{ when: new Date(0), n: NaN }])
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual(["(index)\twhen\tn\n0\t1970-01-01T00:00:00.000Z\tNaN"]) expect(result.logs).toStrictEqual(["(index)\twhen\tn\n0\t1970-01-01T00:00:00.000Z\tNaN"])
}) })
test("captures console.dir and console.table output", async () => { test("captures console.dir and console.table output", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.dir({ nested: { ok: true } }) console.dir({ nested: { ok: true } })
console.table([ console.table([
@@ -309,15 +328,13 @@ describe("CodeMode console capture", () => {
], ["name", "count"]) ], ["name", "count"])
return "done" return "done"
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: "done", value: "done",
logs: [ logs: ['{"nested":{"ok":true}}', "(index)\tname\tcount\n0\tKit\t1\n1\tOlive\t2"],
'{"nested":{"ok":true}}',
"(index)\tname\tcount\n0\tKit\t1\n1\tOlive\t2",
],
toolCalls: [], toolCalls: [],
}) })
}) })
@@ -325,9 +342,11 @@ describe("CodeMode console capture", () => {
describe("CodeMode output budget", () => { describe("CodeMode output budget", () => {
test("absent maxOutputBytes means no truncation at all", async () => { test("absent maxOutputBytes means no truncation at all", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: `console.log("z".repeat(50_000)); return "x".repeat(100_000)`, code: `console.log("z".repeat(50_000)); return "x".repeat(100_000)`,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
@@ -338,29 +357,35 @@ describe("CodeMode output budget", () => {
test("truncates an oversized result value with a marker instead of failing", async () => { test("truncates an oversized result value with a marker instead of failing", async () => {
const limits: ExecutionLimits = { maxOutputBytes: 40 } const limits: ExecutionLimits = { maxOutputBytes: 40 }
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: `return { data: "${"x".repeat(200)}" }`, code: `return { data: "${"x".repeat(200)}" }`,
limits, limits,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
expect(result.truncated).toBe(true) expect(result.truncated).toBe(true)
expect(typeof result.value).toBe("string") expect(typeof result.value).toBe("string")
expect(result.value).toMatch(/^\{"data":"x+ \[result truncated: \d+ bytes exceeds the 40-byte output limit; return a smaller value\]$/) expect(result.value).toMatch(
/^\{"data":"x+ \[result truncated: \d+ bytes exceeds the 40-byte output limit; return a smaller value\]$/,
)
expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(result)))).toStrictEqual(result) expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(result)))).toStrictEqual(result)
}) })
test("keeps leading logs within the remaining budget and marks the cut", async () => { test("keeps leading logs within the remaining budget and marks the cut", async () => {
const limits: ExecutionLimits = { maxOutputBytes: 40 } const limits: ExecutionLimits = { maxOutputBytes: 40 }
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log("first line") console.log("first line")
console.log("${"y".repeat(200)}") console.log("${"y".repeat(200)}")
return "ok" return "ok"
`, `,
limits, limits,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
@@ -370,12 +395,14 @@ describe("CodeMode output budget", () => {
}) })
test("does not mark results within the budget", async () => { test("does not mark results within the budget", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: ` code: `
console.log("fits") console.log("fits")
return { fits: true } return { fits: true }
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: { fits: true }, value: { fits: true },
@@ -395,18 +422,21 @@ describe("CodeMode schema flexibility", () => {
properties: { id: { type: "string" }, count: { type: "number" } }, properties: { id: { type: "string" }, count: { type: "number" } },
required: ["id"], required: ["id"],
}, },
run: (input) => Effect.sync(() => { run: (input) =>
Effect.sync(() => {
observed.push(input) observed.push(input)
return { echoed: input } return { echoed: input }
}), }),
}) })
const runtime = CodeMode.make({ tools: { adapter: { call } } }) const runtime = CodeMode.make({ tools: { adapter: { call } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "adapter.call", path: "adapter.call",
description: "Call an adapter-described tool", description: "Call an adapter-described tool",
signature: "tools.adapter.call(input: { id: string; count?: number }): Promise<unknown>", signature: "tools.adapter.call(input: { id: string; count?: number }): Promise<unknown>",
}]) },
])
// JSON Schema is render-only: mistyped input passes through unvalidated. // JSON Schema is render-only: mistyped input passes through unvalidated.
const result = await Effect.runPromise(runtime.execute(`return await tools.adapter.call({ id: 42 })`)) const result = await Effect.runPromise(runtime.execute(`return await tools.adapter.call({ id: 42 })`))
@@ -421,17 +451,25 @@ describe("CodeMode schema flexibility", () => {
input: { type: "object", properties: { login: { type: "string" } }, required: ["login"] }, input: { type: "object", properties: { login: { type: "string" } }, required: ["login"] },
output: { output: {
$ref: "#/$defs/User", $ref: "#/$defs/User",
$defs: { User: { type: "object", properties: { login: { type: "string" }, id: { type: "number" } }, required: ["login", "id"] } }, $defs: {
User: {
type: "object",
properties: { login: { type: "string" }, id: { type: "number" } },
required: ["login", "id"],
},
},
}, },
run: () => Effect.succeed({ login: "kit", id: 7 }), run: () => Effect.succeed({ login: "kit", id: 7 }),
}) })
const runtime = CodeMode.make({ tools: { users: { lookup } } }) const runtime = CodeMode.make({ tools: { users: { lookup } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "users.lookup", path: "users.lookup",
description: "Look up a user", description: "Look up a user",
signature: "tools.users.lookup(input: { login: string }): Promise<{ login: string; id: number }>", signature: "tools.users.lookup(input: { login: string }): Promise<{ login: string; id: number }>",
}]) },
])
const result = await Effect.runPromise(runtime.execute(`return await tools.users.lookup({ login: "kit" })`)) const result = await Effect.runPromise(runtime.execute(`return await tools.users.lookup({ login: "kit" })`))
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
@@ -478,16 +516,20 @@ describe("CodeMode public contract", () => {
expect(agentTool.input).toBe(ExecuteInputSchema) expect(agentTool.input).toBe(ExecuteInputSchema)
expect(agentTool.output).toBe(ExecuteResultSchema) expect(agentTool.output).toBe(ExecuteResultSchema)
expect(agentTool.description).toBe(runtime.instructions()) expect(agentTool.description).toBe(runtime.instructions())
expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(projected)))).toStrictEqual(projected) expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(projected)))).toStrictEqual(
projected,
)
}) })
test("inlines a COMPLETE small catalog and keeps search registered but unadvertised", async () => { test("inlines a COMPLETE small catalog and keeps search registered but unadvertised", async () => {
const runtime = CodeMode.make({ tools }) const runtime = CodeMode.make({ tools })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "orders.lookup", path: "orders.lookup",
description: "Look up an order by ID", description: "Look up an order by ID",
signature: "tools.orders.lookup(input: { id: string }): Promise<{ id: string; status: string }>", signature: "tools.orders.lookup(input: { id: string }): Promise<{ id: string; status: string }>",
}]) },
])
expect(runtime.instructions()).toContain("Available tools (COMPLETE list") expect(runtime.instructions()).toContain("Available tools (COMPLETE list")
expect(runtime.instructions()).toContain("- orders (1 tool)") expect(runtime.instructions()).toContain("- orders (1 tool)")
expect(runtime.instructions()).toContain( expect(runtime.instructions()).toContain(
@@ -502,11 +544,13 @@ describe("CodeMode public contract", () => {
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (result.ok) { if (result.ok) {
expect(result.value).toStrictEqual({ expect(result.value).toStrictEqual({
items: [{ items: [
{
path: "tools.orders.lookup", path: "tools.orders.lookup",
description: "Look up an order by ID", description: "Look up an order by ID",
signature: "tools.orders.lookup(input: {\n id: string\n}): Promise<{\n id: string\n status: string\n}>", signature: "tools.orders.lookup(input: {\n id: string\n}): Promise<{\n id: string\n status: string\n}>",
}], },
],
total: 1, total: 1,
}) })
} }
@@ -521,31 +565,43 @@ describe("CodeMode public contract", () => {
}) })
const runtime = CodeMode.make({ tools: { context7: { "resolve-library-id": resolveLibrary } } }) const runtime = CodeMode.make({ tools: { context7: { "resolve-library-id": resolveLibrary } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
{
path: "context7.resolve-library-id", path: "context7.resolve-library-id",
description: "Resolve a library ID", description: "Resolve a library ID",
signature: 'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>', signature: 'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>',
}]) },
expect(runtime.instructions()).toContain('tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>') ])
expect(runtime.instructions()).toContain(
'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>',
)
const search = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: "resolve library id" })`)) const search = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: "resolve library id" })`),
)
expect(search.ok).toBe(true) expect(search.ok).toBe(true)
if (search.ok) { if (search.ok) {
expect(search.value).toStrictEqual({ expect(search.value).toStrictEqual({
items: [{ items: [
{
path: 'tools.context7["resolve-library-id"]', path: 'tools.context7["resolve-library-id"]',
description: "Resolve a library ID", description: "Resolve a library ID",
signature: 'tools.context7["resolve-library-id"](input: {\n libraryName: string\n}): Promise<string>', signature: 'tools.context7["resolve-library-id"](input: {\n libraryName: string\n}): Promise<string>',
}], },
],
total: 1, total: 1,
}) })
} }
const call = await Effect.runPromise(runtime.execute(`return await tools.context7["resolve-library-id"]({ libraryName: "TypeScript" })`)) const call = await Effect.runPromise(
runtime.execute(`return await tools.context7["resolve-library-id"]({ libraryName: "TypeScript" })`),
)
expect(call.ok).toBe(true) expect(call.ok).toBe(true)
if (call.ok) expect(call.value).toBe("/resolved/TypeScript") if (call.ok) expect(call.value).toBe("/resolved/TypeScript")
const exact = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: 'tools.context7["resolve-library-id"]' })`)) const exact = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: 'tools.context7["resolve-library-id"]' })`),
)
expect(exact.ok).toBe(true) expect(exact.ok).toBe(true)
if (exact.ok) expect((exact.value as { total: number }).total).toBe(1) if (exact.ok) expect((exact.value as { total: number }).total).toBe(1)
}) })
@@ -561,7 +617,9 @@ describe("CodeMode public contract", () => {
expect(instructions.indexOf("## Rules")).toBeLessThan(instructions.indexOf("## Syntax")) expect(instructions.indexOf("## Rules")).toBeLessThan(instructions.indexOf("## Syntax"))
expect(instructions.indexOf("## Syntax")).toBeLessThan(instructions.indexOf("\n## Available tools (COMPLETE list")) expect(instructions.indexOf("## Syntax")).toBeLessThan(instructions.indexOf("\n## Available tools (COMPLETE list"))
// The workflow carries the result-shape guidance; Rules only add content beyond it. // The workflow carries the result-shape guidance; Rules only add content beyond it.
expect(instructions).toContain('`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string') expect(instructions).toContain(
'`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string',
)
expect(instructions).toContain("Return only the fields you need") expect(instructions).toContain("Return only the fields you need")
expect(instructions).toContain("raw payloads get truncated and waste context") expect(instructions).toContain("raw payloads get truncated and waste context")
expect(instructions).toContain("`const res = await tools.<namespace>.<tool>(input)`") expect(instructions).toContain("`const res = await tools.<namespace>.<tool>(input)`")
@@ -584,8 +642,12 @@ describe("CodeMode public contract", () => {
expect(partial).toContain( expect(partial).toContain(
'1. Find a tool (skip when it is already listed below): `const { items } = await tools.$codemode.search({ query: "<intent + key nouns>" })` — short phrases like "list issues" work best.', '1. Find a tool (skip when it is already listed below): `const { items } = await tools.$codemode.search({ query: "<intent + key nouns>" })` — short phrases like "list issues" work best.',
) )
expect(partial).toContain("Only tools listed here or returned by `tools.$codemode.search` are available inside `tools`") expect(partial).toContain(
expect(partial).toContain('- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.') "Only tools listed here or returned by `tools.$codemode.search` are available inside `tools`",
)
expect(partial).toContain(
'- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.',
)
expect(partial).not.toContain("total_count") expect(partial).not.toContain("total_count")
expect(partial).not.toContain("tools.orders.lookup({") expect(partial).not.toContain("tools.orders.lookup({")
}) })
@@ -604,7 +666,9 @@ describe("CodeMode public contract", () => {
expect(instructions).not.toContain("instanceof Error") expect(instructions).not.toContain("instanceof Error")
expect(instructions).not.toContain("splice") expect(instructions).not.toContain("splice")
// The data-boundary note survives. // The data-boundary note survives.
expect(instructions).toContain("Dates serialize to ISO strings at data boundaries; Map/Set/RegExp serialize to `{}`.") expect(instructions).toContain(
"Dates serialize to ISO strings at data boundaries; Map/Set/RegExp serialize to `{}`.",
)
}) })
test("zero tools keep minimal sections and the no-tools notice", () => { test("zero tools keep minimal sections and the no-tools notice", () => {
@@ -635,18 +699,22 @@ describe("CodeMode public contract", () => {
tools: { thread: { uploadFile: upload, generateImage: generate }, orders: { lookup } }, tools: { thread: { uploadFile: upload, generateImage: generate }, orders: { lookup } },
discovery: { maxInlineCatalogTokens: 0 }, discovery: { maxInlineCatalogTokens: 0 },
}) })
expect(runtime.instructions()).toContain("Available tools (PARTIAL — 0 of 3 shown; find the rest with tools.$codemode.search)") expect(runtime.instructions()).toContain(
"Available tools (PARTIAL — 0 of 3 shown; find the rest with tools.$codemode.search)",
)
expect(runtime.instructions()).toContain("- thread (2 tools, none shown)") expect(runtime.instructions()).toContain("- thread (2 tools, none shown)")
expect(runtime.instructions()).toContain("- orders (1 tool, none shown)") expect(runtime.instructions()).toContain("- orders (1 tool, none shown)")
expect(runtime.instructions()).toMatch(/\$codemode\.search/) expect(runtime.instructions()).toMatch(/\$codemode\.search/)
expect(runtime.instructions()).not.toMatch(/tools\.thread\.uploadFile\(input/) expect(runtime.instructions()).not.toMatch(/tools\.thread\.uploadFile\(input/)
const result = await Effect.runPromise(runtime.execute(` const result = await Effect.runPromise(
runtime.execute(`
return await tools.$codemode.search({ return await tools.$codemode.search({
query: "send message attachment upload file to current Discord thread", query: "send message attachment upload file to current Discord thread",
limit: 2 limit: 2
}) })
`)) `),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
expect(result.value).toStrictEqual({ expect(result.value).toStrictEqual({
@@ -666,19 +734,27 @@ describe("CodeMode public contract", () => {
}) })
expect(result.toolCalls).toStrictEqual([{ name: "$codemode.search" }]) expect(result.toolCalls).toStrictEqual([{ name: "$codemode.search" }])
const variants = await Effect.runPromise(runtime.execute(` const variants = await Effect.runPromise(
runtime.execute(`
return await Promise.all([ return await Promise.all([
tools.$codemode.search({ query: "file" }), tools.$codemode.search({ query: "file" }),
tools.$codemode.search({ query: "image" }) tools.$codemode.search({ query: "image" })
]) ])
`)) `),
)
expect(variants.ok).toBe(true) expect(variants.ok).toBe(true)
if (variants.ok) { if (variants.ok) {
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[0]?.items[0]?.path).toBe("tools.thread.uploadFile") expect((variants.value as Array<{ items: Array<{ path: string }> }>)[0]?.items[0]?.path).toBe(
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[1]?.items[0]?.path).toBe("tools.thread.generateImage") "tools.thread.uploadFile",
)
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[1]?.items[0]?.path).toBe(
"tools.thread.generateImage",
)
} }
const removed = await Effect.runPromise(runtime.execute(`return await tools.$codemode.describe({ path: "thread.uploadFile" })`)) const removed = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.describe({ path: "thread.uploadFile" })`),
)
expect(removed.ok).toBe(false) expect(removed.ok).toBe(false)
if (!removed.ok) expect(removed.error.kind).toBe("UnknownTool") if (!removed.ok) expect(removed.error.kind).toBe("UnknownTool")
}) })
@@ -706,15 +782,19 @@ describe("CodeMode public contract", () => {
} }
for (const query of ["many.tool13", "tools.many.tool13"]) { for (const query of ["many.tool13", "tools.many.tool13"]) {
const exact = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`)) const exact = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`),
)
expect(exact.ok).toBe(true) expect(exact.ok).toBe(true)
if (exact.ok) { if (exact.ok) {
expect(exact.value).toStrictEqual({ expect(exact.value).toStrictEqual({
items: [{ items: [
{
path: "tools.many.tool13", path: "tools.many.tool13",
description: "Numbered tool 13", description: "Numbered tool 13",
signature: "tools.many.tool13(input: {\n id: string\n}): Promise<string>", signature: "tools.many.tool13(input: {\n id: string\n}): Promise<string>",
}], },
],
total: 1, total: 1,
}) })
} }
@@ -737,20 +817,23 @@ describe("CodeMode public contract", () => {
}) })
// Empty query + namespace browses just that namespace, alphabetical by path. // Empty query + namespace browses just that namespace, alphabetical by path.
const browse = await Effect.runPromise(runtime.execute( const browse = await Effect.runPromise(
`return await tools.$codemode.search({ query: "", namespace: "github" })`, runtime.execute(`return await tools.$codemode.search({ query: "", namespace: "github" })`),
)) )
expect(browse.ok).toBe(true) expect(browse.ok).toBe(true)
if (browse.ok) { if (browse.ok) {
const value = browse.value as { items: Array<{ path: string }>; total: number } const value = browse.value as { items: Array<{ path: string }>; total: number }
expect(value.total).toBe(2) expect(value.total).toBe(2)
expect(value.items.map((item) => item.path)).toStrictEqual(["tools.github.create_issue", "tools.github.list_issues"]) expect(value.items.map((item) => item.path)).toStrictEqual([
"tools.github.create_issue",
"tools.github.list_issues",
])
} }
// A query + namespace ranks within that namespace only. // A query + namespace ranks within that namespace only.
const scoped = await Effect.runPromise(runtime.execute( const scoped = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: "linear" })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: "linear" })`),
)) )
expect(scoped.ok).toBe(true) expect(scoped.ok).toBe(true)
if (scoped.ok) { if (scoped.ok) {
const value = scoped.value as { items: Array<{ path: string }>; total: number } const value = scoped.value as { items: Array<{ path: string }>; total: number }
@@ -758,9 +841,9 @@ describe("CodeMode public contract", () => {
expect(value.items[0]?.path).toBe("tools.linear.list_issues") expect(value.items[0]?.path).toBe("tools.linear.list_issues")
} }
const invalid = await Effect.runPromise(runtime.execute( const invalid = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: 7 })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: 7 })`),
)) )
expect(invalid.ok).toBe(false) expect(invalid.ok).toBe(false)
if (!invalid.ok) expect(invalid.error.kind).toBe("InvalidToolInput") if (!invalid.ok) expect(invalid.error.kind).toBe("InvalidToolInput")
}) })
@@ -785,9 +868,9 @@ describe("CodeMode public contract", () => {
// "attachment" appears in neither path nor description — only in the input schema's // "attachment" appears in neither path nor description — only in the input schema's
// property names, which the searchable text includes. // property names, which the searchable text includes.
const byParameter = await Effect.runPromise(runtime.execute( const byParameter = await Effect.runPromise(
`return await tools.$codemode.search({ query: "attachment" })`, runtime.execute(`return await tools.$codemode.search({ query: "attachment" })`),
)) )
expect(byParameter.ok).toBe(true) expect(byParameter.ok).toBe(true)
if (byParameter.ok) { if (byParameter.ok) {
const value = byParameter.value as { items: Array<{ path: string }>; total: number } const value = byParameter.value as { items: Array<{ path: string }>; total: number }
@@ -796,9 +879,9 @@ describe("CodeMode public contract", () => {
} }
// Substring matching: a partial word ("docum") still hits the description. // Substring matching: a partial word ("docum") still hits the description.
const bySubstring = await Effect.runPromise(runtime.execute( const bySubstring = await Effect.runPromise(
`return await tools.$codemode.search({ query: "docum" })`, runtime.execute(`return await tools.$codemode.search({ query: "docum" })`),
)) )
expect(bySubstring.ok).toBe(true) expect(bySubstring.ok).toBe(true)
if (bySubstring.ok) { if (bySubstring.ok) {
const value = bySubstring.value as { items: Array<{ path: string }>; total: number } const value = bySubstring.value as { items: Array<{ path: string }>; total: number }
@@ -825,9 +908,9 @@ describe("CodeMode public contract", () => {
}) })
// "issues" still finds the singular-only tool (term OR singular(term) per field)... // "issues" still finds the singular-only tool (term OR singular(term) per field)...
const plural = await Effect.runPromise(runtime.execute( const plural = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: "tracker" })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: "tracker" })`),
)) )
expect(plural.ok).toBe(true) expect(plural.ok).toBe(true)
if (plural.ok) { if (plural.ok) {
const value = plural.value as { items: Array<{ path: string }>; total: number } const value = plural.value as { items: Array<{ path: string }>; total: number }
@@ -836,14 +919,15 @@ describe("CodeMode public contract", () => {
} }
// ...while a true "issues" path match still outranks the singular-only description match. // ...while a true "issues" path match still outranks the singular-only description match.
const ranked = await Effect.runPromise(runtime.execute( const ranked = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: "issues" })`))
`return await tools.$codemode.search({ query: "issues" })`,
))
expect(ranked.ok).toBe(true) expect(ranked.ok).toBe(true)
if (ranked.ok) { if (ranked.ok) {
const value = ranked.value as { items: Array<{ path: string }>; total: number } const value = ranked.value as { items: Array<{ path: string }>; total: number }
expect(value.total).toBe(2) expect(value.total).toBe(2)
expect(value.items.map((item) => item.path)).toStrictEqual(["tools.github.list_issues", "tools.tracker.fetch_all"]) expect(value.items.map((item) => item.path)).toStrictEqual([
"tools.github.list_issues",
"tools.tracker.fetch_all",
])
} }
}) })
@@ -882,8 +966,12 @@ describe("CodeMode public contract", () => {
run: () => Effect.succeed("ok"), run: () => Effect.succeed("ok"),
}) })
const expensive = Tool.make({ const expensive = Tool.make({
description: "An expensive tool whose description alone consumes far more than the remaining inline catalog byte budget for this runtime", description:
input: Schema.Struct({ someRatherLongParameterName: Schema.String, anotherEvenLongerParameterName: Schema.Number }), "An expensive tool whose description alone consumes far more than the remaining inline catalog byte budget for this runtime",
input: Schema.Struct({
someRatherLongParameterName: Schema.String,
anotherEvenLongerParameterName: Schema.Number,
}),
output: Schema.String, output: Schema.String,
run: () => Effect.succeed("ok"), run: () => Effect.succeed("ok"),
}) })
@@ -896,7 +984,9 @@ describe("CodeMode public contract", () => {
}) })
const instructions = runtime.instructions() const instructions = runtime.instructions()
expect(instructions).toContain("Available tools (PARTIAL — 2 of 3 shown; find the rest with tools.$codemode.search)") expect(instructions).toContain(
"Available tools (PARTIAL — 2 of 3 shown; find the rest with tools.$codemode.search)",
)
expect(instructions).toContain("- alpha (2 tools, 1 shown)") expect(instructions).toContain("- alpha (2 tools, 1 shown)")
expect(instructions).toContain(" - tools.alpha.cheap(input: { q: string }): Promise<string> // Cheap") expect(instructions).toContain(" - tools.alpha.cheap(input: { q: string }): Promise<string> // Cheap")
expect(instructions).not.toContain("tools.alpha.expensive(") expect(instructions).not.toContain("tools.alpha.expensive(")
@@ -912,7 +1002,8 @@ describe("CodeMode public contract", () => {
description: "Double a number", description: "Double a number",
input: Schema.Struct({ value: Schema.NumberFromString }), input: Schema.Struct({ value: Schema.NumberFromString }),
output: Schema.NumberFromString, output: Schema.NumberFromString,
run: ({ value }) => Effect.sync(() => { run: ({ value }) =>
Effect.sync(() => {
observed.push(value) observed.push(value)
return String(value * 2) return String(value * 2)
}), }),
@@ -934,9 +1025,11 @@ describe("CodeMode public contract", () => {
}) })
test("returns JSON-safe data and normalizes undefined to null", async () => { test("returns JSON-safe data and normalizes undefined to null", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
code: `return { top: undefined, nested: [1, undefined] }`, code: `return { top: undefined, nested: [1, undefined] }`,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: { top: null, nested: [1, null] }, value: { top: null, nested: [1, null] },
@@ -947,18 +1040,20 @@ describe("CodeMode public contract", () => {
test("rejects invalid configuration and discovery limits", async () => { test("rejects invalid configuration and discovery limits", async () => {
expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: 0 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: 0 } })).toThrow(RangeError)
expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: Number.POSITIVE_INFINITY } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: Number.POSITIVE_INFINITY } })).toThrow(
RangeError,
)
expect(() => CodeMode.execute({ code: "return 1", limits: { maxToolCalls: -1 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { maxToolCalls: -1 } })).toThrow(RangeError)
expect(() => CodeMode.execute({ code: "return 1", limits: { maxOutputBytes: -1 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { maxOutputBytes: -1 } })).toThrow(RangeError)
expect(() => CodeMode.make({ tools, discovery: { maxInlineCatalogTokens: -1 } })).toThrow(RangeError) expect(() => CodeMode.make({ tools, discovery: { maxInlineCatalogTokens: -1 } })).toThrow(RangeError)
const result = await Effect.runPromise(CodeMode.make({ const result = await Effect.runPromise(
CodeMode.make({
tools, tools,
discovery: { maxInlineCatalogTokens: 0 }, discovery: { maxInlineCatalogTokens: 0 },
}).execute( }).execute(`return await tools.$codemode.search({ query: "order", limit: 0.5 })`),
`return await tools.$codemode.search({ query: "order", limit: 0.5 })`, )
))
expect(result.ok).toBe(false) expect(result.ok).toBe(false)
if (result.ok) return if (result.ok) return
expect(result.error.kind).toBe("InvalidToolInput") expect(result.error.kind).toBe("InvalidToolInput")
@@ -979,14 +1074,16 @@ describe("CodeMode public contract", () => {
output: Schema.Number, output: Schema.Number,
run: () => Effect.succeed(1), run: () => Effect.succeed(1),
}) })
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
tools: { host: { count: counter } }, tools: { host: { count: counter } },
code: ` code: `
let total = 0 let total = 0
for (let i = 0; i < 150; i += 1) total += await tools.host.count({}) for (let i = 0; i < 150; i += 1) total += await tools.host.count({})
return total return total
`, `,
})) }),
)
expect(result).toMatchObject({ ok: true, value: 150 }) expect(result).toMatchObject({ ok: true, value: 150 })
if (result.ok) expect(result.toolCalls.length).toBe(150) if (result.ok) expect(result.toolCalls.length).toBe(150)
}) })
@@ -1008,8 +1105,6 @@ describe("CodeMode public contract", () => {
}) })
test("reserves the discovery namespace", () => { test("reserves the discovery namespace", () => {
expect(() => CodeMode.make({ tools: { $codemode: { lookup } } })).toThrow( expect(() => CodeMode.make({ tools: { $codemode: { lookup } } })).toThrow(/reserved for CodeMode discovery tools/)
/reserved for CodeMode discovery tools/,
)
}) })
}) })
+24 -12
View File
@@ -36,10 +36,12 @@ const error = async (code: string) => {
describe("Object.keys over tool references", () => { describe("Object.keys over tool references", () => {
test("enumerates top-level namespaces (the transcript program)", async () => { test("enumerates top-level namespaces (the transcript program)", async () => {
expect(await value(` expect(
await value(`
const namespaces = Object.keys(tools) const namespaces = Object.keys(tools)
return { namespaces, count: namespaces.length } return { namespaces, count: namespaces.length }
`)).toEqual({ namespaces: ["github", "memory", "playwright"], count: 3 }) `),
).toEqual({ namespaces: ["github", "memory", "playwright"], count: 3 })
}) })
test("enumerates tool names at a nested namespace", async () => { test("enumerates tool names at a nested namespace", async () => {
@@ -92,7 +94,8 @@ describe("Object.keys over arrays", () => {
describe("for...in", () => { describe("for...in", () => {
test("iterates own enumerable keys of a plain object with break/continue", async () => { test("iterates own enumerable keys of a plain object with break/continue", async () => {
expect(await value(` expect(
await value(`
const seen = [] const seen = []
for (const key in { a: 1, b: 2, c: 3, d: 4 }) { for (const key in { a: 1, b: 2, c: 3, d: 4 }) {
if (key === "b") continue if (key === "b") continue
@@ -100,41 +103,50 @@ describe("for...in", () => {
seen.push(key) seen.push(key)
} }
return seen return seen
`)).toEqual(["a", "c"]) `),
).toEqual(["a", "c"])
}) })
test("iterates index strings over arrays", async () => { test("iterates index strings over arrays", async () => {
expect(await value(` expect(
await value(`
const indexes = [] const indexes = []
for (const i in ["x", "y", "z"]) { for (const i in ["x", "y", "z"]) {
if (i === "2") break if (i === "2") break
indexes.push(i) indexes.push(i)
} }
return indexes return indexes
`)).toEqual(["0", "1"]) `),
).toEqual(["0", "1"])
}) })
test("supports let declarations and bare identifiers", async () => { test("supports let declarations and bare identifiers", async () => {
expect(await value(` expect(
await value(`
let last = "" let last = ""
for (let key in { a: 1, b: 2 }) last = key for (let key in { a: 1, b: 2 }) last = key
return last return last
`)).toBe("b") `),
expect(await value(` ).toBe("b")
expect(
await value(`
let key = "before" let key = "before"
for (key in { only: 1 }) {} for (key in { only: 1 }) {}
return key return key
`)).toBe("only") `),
).toBe("only")
}) })
test("enumerates namespaces and tools from the host tool tree", async () => { test("enumerates namespaces and tools from the host tool tree", async () => {
expect(await value(` expect(
await value(`
const names = [] const names = []
for (const ns in tools) { for (const ns in tools) {
for (const name in tools[ns]) names.push(ns + "." + name) for (const name in tools[ns]) names.push(ns + "." + name)
} }
return names return names
`)).toEqual(["github.list_issues", "github.get_issue", "memory.search", "playwright.navigate"]) `),
).toEqual(["github.list_issues", "github.get_issue", "memory.search", "playwright.navigate"])
}) })
test("unsupported values fail with a hint at for...of and Object.keys", async () => { test("unsupported values fail with a hint at for...of and Object.keys", async () => {
+95 -50
View File
@@ -143,21 +143,40 @@ describe("H1: NaN/Infinity flow as intermediates and normalize to null at the bo
describe("Error values and instanceof", () => { describe("Error values and instanceof", () => {
test("new Error carries name/message and is instanceof Error", async () => { test("new Error carries name/message and is instanceof Error", async () => {
expect(await value(`const e = new Error("boom"); return [e instanceof Error, e.name, e.message]`)).toEqual([true, "Error", "boom"]) expect(await value(`const e = new Error("boom"); return [e instanceof Error, e.name, e.message]`)).toEqual([
true,
"Error",
"boom",
])
}) })
test("Error without new behaves like new Error", async () => { test("Error without new behaves like new Error", async () => {
expect(await value(`const e = Error("plain"); return [e instanceof Error, e.name, e.message]`)).toEqual([true, "Error", "plain"]) expect(await value(`const e = Error("plain"); return [e instanceof Error, e.name, e.message]`)).toEqual([
expect(await value(`const e = new Error(); return [e.name, e.message, e instanceof Error]`)).toEqual(["Error", "", true]) true,
"Error",
"plain",
])
expect(await value(`const e = new Error(); return [e.name, e.message, e instanceof Error]`)).toEqual([
"Error",
"",
true,
])
}) })
test("specific error types are instanceof themselves and Error, not each other", async () => { test("specific error types are instanceof themselves and Error, not each other", async () => {
expect(await value(`const e = new TypeError("t"); return [e instanceof TypeError, e instanceof Error, e instanceof RangeError]`)).toEqual([true, true, false]) expect(
await value(
`const e = new TypeError("t"); return [e instanceof TypeError, e instanceof Error, e instanceof RangeError]`,
),
).toEqual([true, true, false])
expect(await value(`return new Error("e") instanceof TypeError`)).toBe(false) expect(await value(`return new Error("e") instanceof TypeError`)).toBe(false)
}) })
test("thrown errors keep instanceof through try/catch", async () => { test("thrown errors keep instanceof through try/catch", async () => {
expect(await value(`try { throw new Error("x") } catch (e) { return [e instanceof Error, e.message] }`)).toEqual([true, "x"]) expect(await value(`try { throw new Error("x") } catch (e) { return [e instanceof Error, e.message] }`)).toEqual([
true,
"x",
])
}) })
test("interpreter runtime failures are caught as Error values", async () => { test("interpreter runtime failures are caught as Error values", async () => {
@@ -168,33 +187,48 @@ describe("Error values and instanceof", () => {
test("caught failures carry the constructor name the real-JS failure would have", async () => { test("caught failures carry the constructor name the real-JS failure would have", async () => {
// JSON.parse throws SyntaxError: name and specific-instanceof both carry through, and the // JSON.parse throws SyntaxError: name and specific-instanceof both carry through, and the
// message keeps the engine's position detail. // message keeps the engine's position detail.
expect(await value(` expect(
await value(`
try { JSON.parse("{oops") } catch (e) { try { JSON.parse("{oops") } catch (e) {
return [e.name, e instanceof SyntaxError, e instanceof Error, e instanceof TypeError, e.message.includes("JSON")] return [e.name, e instanceof SyntaxError, e instanceof Error, e instanceof TypeError, e.message.includes("JSON")]
} }
`)).toEqual(["SyntaxError", true, true, false, true]) `),
expect(await value(`try { undeclared() } catch (e) { return [e.name, e instanceof ReferenceError] }`)) ).toEqual(["SyntaxError", true, true, false, true])
.toEqual(["ReferenceError", true]) expect(await value(`try { undeclared() } catch (e) { return [e.name, e instanceof ReferenceError] }`)).toEqual([
expect(await value(`try { const c = 1; c = 2 } catch (e) { return [e.name, e instanceof TypeError] }`)) "ReferenceError",
.toEqual(["TypeError", true]) true,
expect(await value(`try { "a".normalize("NOPE") } catch (e) { return [e.name, e instanceof RangeError] }`)) ])
.toEqual(["RangeError", true]) expect(await value(`try { const c = 1; c = 2 } catch (e) { return [e.name, e instanceof TypeError] }`)).toEqual([
expect(await value(`try { "a".match("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)) "TypeError",
.toEqual(["SyntaxError", true]) true,
expect(await value(`try { new RegExp("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)) ])
.toEqual(["SyntaxError", true]) expect(await value(`try { "a".normalize("NOPE") } catch (e) { return [e.name, e instanceof RangeError] }`)).toEqual(
["RangeError", true],
)
expect(await value(`try { "a".match("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)).toEqual([
"SyntaxError",
true,
])
expect(await value(`try { new RegExp("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)).toEqual([
"SyntaxError",
true,
])
}) })
test("diagnostics without a specific real-JS analogue are named plain Error", async () => { test("diagnostics without a specific real-JS analogue are named plain Error", async () => {
expect(await value(`try { JSON.parse(5) } catch (e) { return [e.name, e instanceof Error] }`)) expect(await value(`try { JSON.parse(5) } catch (e) { return [e.name, e instanceof Error] }`)).toEqual([
.toEqual(["Error", true]) "Error",
true,
])
}) })
test("Promise.allSettled rejection reasons are Error values", async () => { test("Promise.allSettled rejection reasons are Error values", async () => {
expect(await value(` expect(
await value(`
const settled = await Promise.allSettled([Promise.reject(new Error("b"))]) const settled = await Promise.allSettled([Promise.reject(new Error("b"))])
return [settled[0].reason instanceof Error, settled[0].reason.message] return [settled[0].reason instanceof Error, settled[0].reason.message]
`)).toEqual([true, "b"]) `),
).toEqual([true, "b"])
}) })
test("non-error thrown values are not instanceof Error", async () => { test("non-error thrown values are not instanceof Error", async () => {
@@ -203,7 +237,11 @@ describe("Error values and instanceof", () => {
}) })
test("plain data is never instanceof Error", async () => { test("plain data is never instanceof Error", async () => {
expect(await value(`return [({}) instanceof Error, "s" instanceof Error, null instanceof Error]`)).toEqual([false, false, false]) expect(await value(`return [({}) instanceof Error, "s" instanceof Error, null instanceof Error]`)).toEqual([
false,
false,
false,
])
}) })
test("error values still serialize as plain { name, message } data", async () => { test("error values still serialize as plain { name, message } data", async () => {
@@ -227,17 +265,29 @@ describe("Error values and instanceof", () => {
describe("array methods: splice, fill, copyWithin, keys/values/entries", () => { describe("array methods: splice, fill, copyWithin, keys/values/entries", () => {
test("splice removes in place and returns the removed elements", async () => { test("splice removes in place and returns the removed elements", async () => {
expect(await value(`const a = [1,2,3,4]; const removed = a.splice(1, 2); return { removed, a }`)).toEqual({ removed: [2, 3], a: [1, 4] }) expect(await value(`const a = [1,2,3,4]; const removed = a.splice(1, 2); return { removed, a }`)).toEqual({
removed: [2, 3],
a: [1, 4],
})
}) })
test("splice inserts new elements at the cut", async () => { test("splice inserts new elements at the cut", async () => {
expect(await value(`const a = ["a","d"]; a.splice(1, 0, "b", "c"); return a`)).toEqual(["a", "b", "c", "d"]) expect(await value(`const a = ["a","d"]; a.splice(1, 0, "b", "c"); return a`)).toEqual(["a", "b", "c", "d"])
expect(await value(`const a = [1,2,3]; const removed = a.splice(1, 1, "x"); return { removed, a }`)).toEqual({ removed: [2], a: [1, "x", 3] }) expect(await value(`const a = [1,2,3]; const removed = a.splice(1, 1, "x"); return { removed, a }`)).toEqual({
removed: [2],
a: [1, "x", 3],
})
}) })
test("splice with one argument removes to the end; negative start counts back", async () => { test("splice with one argument removes to the end; negative start counts back", async () => {
expect(await value(`const a = [1,2,3]; const removed = a.splice(1); return { removed, a }`)).toEqual({ removed: [2, 3], a: [1] }) expect(await value(`const a = [1,2,3]; const removed = a.splice(1); return { removed, a }`)).toEqual({
expect(await value(`const a = [1,2,3]; const removed = a.splice(-1); return { removed, a }`)).toEqual({ removed: [3], a: [1, 2] }) removed: [2, 3],
a: [1],
})
expect(await value(`const a = [1,2,3]; const removed = a.splice(-1); return { removed, a }`)).toEqual({
removed: [3],
a: [1, 2],
})
}) })
test("splice rejects inserting a container into itself", async () => { test("splice rejects inserting a container into itself", async () => {
@@ -258,11 +308,13 @@ describe("array methods: splice, fill, copyWithin, keys/values/entries", () => {
test("keys/values/entries return arrays usable with for...of and spread", async () => { test("keys/values/entries return arrays usable with for...of and spread", async () => {
expect(await value(`return [...["x","y","z"].keys()]`)).toEqual([0, 1, 2]) expect(await value(`return [...["x","y","z"].keys()]`)).toEqual([0, 1, 2])
expect(await value(`return ["x","y"].values()`)).toEqual(["x", "y"]) expect(await value(`return ["x","y"].values()`)).toEqual(["x", "y"])
expect(await value(` expect(
await value(`
const out = [] const out = []
for (const [index, item] of ["a","b"].entries()) out.push(index + ":" + item) for (const [index, item] of ["a","b"].entries()) out.push(index + ":" + item)
return out return out
`)).toEqual(["0:a", "1:b"]) `),
).toEqual(["0:a", "1:b"])
expect(await value(`return [...[7].entries()]`)).toEqual([[0, 7]]) expect(await value(`return [...[7].entries()]`)).toEqual([[0, 7]])
}) })
}) })
@@ -300,40 +352,33 @@ describe("compound assignment matches its binary operator", () => {
} }
test("sandbox Date += concatenates its string form, like d = d + 1", async () => { test("sandbox Date += concatenates its string form, like d = d + 1", async () => {
const result = await pair( const result = await pair(`let d = new Date(1000); d += 1; return d`, `let d = new Date(1000); d = d + 1; return d`)
`let d = new Date(1000); d += 1; return d`,
`let d = new Date(1000); d = d + 1; return d`,
)
expect(result).toBe("1970-01-01T00:00:01.000Z1") expect(result).toBe("1970-01-01T00:00:01.000Z1")
}) })
test("sandbox Date numeric compound ops use its time value", async () => { test("sandbox Date numeric compound ops use its time value", async () => {
expect(await pair( expect(
`let d = new Date(1000); d -= 400; return d`, await pair(`let d = new Date(1000); d -= 400; return d`, `let d = new Date(1000); d = d - 400; return d`),
`let d = new Date(1000); d = d - 400; return d`, ).toBe(600)
)).toBe(600) expect(await pair(`let d = new Date(1000); d /= 4; return d`, `let d = new Date(1000); d = d / 4; return d`)).toBe(
expect(await pair( 250,
`let d = new Date(1000); d /= 4; return d`, )
`let d = new Date(1000); d = d / 4; return d`,
)).toBe(250)
}) })
test("string += object/array matches x = x + obj", async () => { test("string += object/array matches x = x + obj", async () => {
expect(await pair( expect(await pair(`let x = "a"; x += { b: 1 }; return x`, `let x = "a"; x = x + { b: 1 }; return x`)).toBe(
`let x = "a"; x += { b: 1 }; return x`, "a[object Object]",
`let x = "a"; x = x + { b: 1 }; return x`, )
)).toBe("a[object Object]") expect(await pair(`let x = "a"; x += [1, 2]; return x`, `let x = "a"; x = x + [1, 2]; return x`)).toBe("a1,2")
expect(await pair(
`let x = "a"; x += [1, 2]; return x`,
`let x = "a"; x = x + [1, 2]; return x`,
)).toBe("a1,2")
}) })
test("compound assignment through a member target coerces the same way", async () => { test("compound assignment through a member target coerces the same way", async () => {
expect(await pair( expect(
await pair(
`const o = { s: "t" }; o.s += new Date(0); return o.s`, `const o = { s: "t" }; o.s += new Date(0); return o.s`,
`const o = { s: "t" }; o.s = o.s + new Date(0); return o.s`, `const o = { s: "t" }; o.s = o.s + new Date(0); return o.s`,
)).toBe("t1970-01-01T00:00:00.000Z") ),
).toBe("t1970-01-01T00:00:00.000Z")
}) })
test("numeric and string compound operators sweep identically to their expansions", async () => { test("numeric and string compound operators sweep identically to their expansions", async () => {
+49 -22
View File
@@ -23,7 +23,7 @@ const sleepyTool = (trace: Trace) =>
input: Schema.Struct({ id: Schema.Number, ms: Schema.optionalKey(Schema.Number) }), input: Schema.Struct({ id: Schema.Number, ms: Schema.optionalKey(Schema.Number) }),
output: Schema.Number, output: Schema.Number,
run: ({ id, ms }) => run: ({ id, ms }) =>
Effect.gen(function*() { Effect.gen(function* () {
trace.starts.push(id) trace.starts.push(id)
trace.active += 1 trace.active += 1
trace.maxActive = Math.max(trace.maxActive, trace.active) trace.maxActive = Math.max(trace.maxActive, trace.active)
@@ -31,10 +31,14 @@ const sleepyTool = (trace: Trace) =>
trace.active -= 1 trace.active -= 1
trace.completed += 1 trace.completed += 1
return id return id
}).pipe(Effect.onInterrupt(() => Effect.sync(() => { }).pipe(
Effect.onInterrupt(() =>
Effect.sync(() => {
trace.active -= 1 trace.active -= 1
trace.interrupted += 1 trace.interrupted += 1
}))), }),
),
),
}) })
const failingTool = Tool.make({ const failingTool = Tool.make({
@@ -46,11 +50,13 @@ const failingTool = Tool.make({
const run = (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}): Promise<ExecuteResult> => { const run = (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}): Promise<ExecuteResult> => {
const trace = options.trace ?? makeTrace() const trace = options.trace ?? makeTrace()
return Effect.runPromise(CodeMode.execute({ return Effect.runPromise(
CodeMode.execute({
tools: { host: { sleepy: sleepyTool(trace), fail: failingTool } }, tools: { host: { sleepy: sleepyTool(trace), fail: failingTool } },
code, code,
...(options.limits ? { limits: options.limits } : {}), ...(options.limits ? { limits: options.limits } : {}),
})) }),
)
} }
const value = async (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}) => { const value = async (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}) => {
@@ -121,7 +127,8 @@ describe("first-class promise values", () => {
}) })
test("an awaited failure is catchable exactly like a synchronous throw", async () => { test("an awaited failure is catchable exactly like a synchronous throw", async () => {
expect(await value(` expect(
await value(`
const p = tools.host.fail({}) const p = tools.host.fail({})
try { try {
await p await p
@@ -129,7 +136,8 @@ describe("first-class promise values", () => {
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a fire-and-forget call completes before the execution ends", async () => { test("a fire-and-forget call completes before the execution ends", async () => {
@@ -186,20 +194,24 @@ describe("promises at data boundaries", () => {
describe("Promise.all over arbitrary arrays", () => { describe("Promise.all over arbitrary arrays", () => {
test("mixes promises and plain values, preserving order", async () => { test("mixes promises and plain values, preserving order", async () => {
expect(await value(` expect(
await value(`
return await Promise.all([tools.host.sleepy({ id: 1 }), "plain", tools.host.sleepy({ id: 2 }), 42]) return await Promise.all([tools.host.sleepy({ id: 1 }), "plain", tools.host.sleepy({ id: 2 }), 42])
`)).toEqual([1, "plain", 2, 42]) `),
).toEqual([1, "plain", 2, 42])
}) })
test("accepts arrays built beforehand, passed as identifiers, and spread elements", async () => { test("accepts arrays built beforehand, passed as identifiers, and spread elements", async () => {
expect(await value(` expect(
await value(`
const calls = [] const calls = []
calls.push(tools.host.sleepy({ id: 1 })) calls.push(tools.host.sleepy({ id: 1 }))
calls.push(7) calls.push(7)
const more = [tools.host.sleepy({ id: 2 })] const more = [tools.host.sleepy({ id: 2 })]
const batch = [...calls, ...more, "x"] const batch = [...calls, ...more, "x"]
return await Promise.all(batch) return await Promise.all(batch)
`)).toEqual([1, 7, 2, "x"]) `),
).toEqual([1, 7, 2, "x"])
}) })
test("runs items.map tool calls in parallel", async () => { test("runs items.map tool calls in parallel", async () => {
@@ -238,14 +250,16 @@ describe("Promise.all over arbitrary arrays", () => {
}) })
test("rejects with the first failure, catchable in-program", async () => { test("rejects with the first failure, catchable in-program", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.all([tools.host.sleepy({ id: 1 }), tools.host.fail({})]) await Promise.all([tools.host.sleepy({ id: 1 }), tools.host.fail({})])
return "no" return "no"
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a non-collection argument is a clear error", async () => { test("a non-collection argument is a clear error", async () => {
@@ -264,14 +278,16 @@ describe("Promise.all over arbitrary arrays", () => {
describe("Promise.allSettled", () => { describe("Promise.allSettled", () => {
test("reports fulfilled and rejected outcomes with catch-normalized reasons", async () => { test("reports fulfilled and rejected outcomes with catch-normalized reasons", async () => {
expect(await value(` expect(
await value(`
return await Promise.allSettled([ return await Promise.allSettled([
tools.host.sleepy({ id: 5 }), tools.host.sleepy({ id: 5 }),
tools.host.fail({}), tools.host.fail({}),
"plain", "plain",
Promise.reject(new Error("boom")), Promise.reject(new Error("boom")),
]) ])
`)).toEqual([ `),
).toEqual([
{ status: "fulfilled", value: 5 }, { status: "fulfilled", value: 5 },
{ status: "rejected", reason: { name: "Error", message: "Lookup refused" } }, { status: "rejected", reason: { name: "Error", message: "Lookup refused" } },
{ status: "fulfilled", value: "plain" }, { status: "fulfilled", value: "plain" },
@@ -306,7 +322,8 @@ describe("Promise.race", () => {
}) })
test("awaiting an interrupted loser afterwards is a catchable program failure", async () => { test("awaiting an interrupted loser afterwards is a catchable program failure", async () => {
expect(await value(` expect(
await value(`
const fast = tools.host.sleepy({ id: 1, ms: 10 }) const fast = tools.host.sleepy({ id: 1, ms: 10 })
const slow = tools.host.sleepy({ id: 2, ms: 5000 }) const slow = tools.host.sleepy({ id: 2, ms: 5000 })
const winner = await Promise.race([fast, slow]) const winner = await Promise.race([fast, slow])
@@ -316,23 +333,31 @@ describe("Promise.race", () => {
} catch (e) { } catch (e) {
return { winner, caught: e.message } return { winner, caught: e.message }
} }
`)).toEqual({ winner: 1, caught: "This tool call was interrupted because another value settled a Promise.race first." }) `),
).toEqual({
winner: 1,
caught: "This tool call was interrupted because another value settled a Promise.race first.",
})
}) })
test("a rejection can win the race", async () => { test("a rejection can win the race", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.race([tools.host.fail({}), tools.host.sleepy({ id: 1, ms: 5000 })]) await Promise.race([tools.host.fail({}), tools.host.sleepy({ id: 1, ms: 5000 })])
return "no" return "no"
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a plain value wins over pending promises", async () => { test("a plain value wins over pending promises", async () => {
const trace = makeTrace() const trace = makeTrace()
expect(await value(`return await Promise.race([tools.host.sleepy({ id: 1, ms: 5000 }), "immediate"])`, { trace })).toBe("immediate") expect(
await value(`return await Promise.race([tools.host.sleepy({ id: 1, ms: 5000 }), "immediate"])`, { trace }),
).toBe("immediate")
expect(trace.interrupted).toBe(1) expect(trace.interrupted).toBe(1)
}) })
@@ -350,14 +375,16 @@ describe("Promise.resolve / Promise.reject", () => {
}) })
test("reject produces a promise whose await throws the reason", async () => { test("reject produces a promise whose await throws the reason", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.reject("nope") await Promise.reject("nope")
return "no" return "no"
} catch (e) { } catch (e) {
return e return e
} }
`)).toBe("nope") `),
).toBe("nope")
}) })
}) })
+19 -14
View File
@@ -83,15 +83,9 @@ describe("pretty signature rendering", () => {
true, true,
) )
expect(pretty).toBe( expect(pretty).toBe(
[ ["{", " /** Search filter */", " filter?: {", " /** Issue state */", " state?: string", " }", "}"].join(
"{", "\n",
" /** Search filter */", ),
" filter?: {",
" /** Issue state */",
" state?: string",
" }",
"}",
].join("\n"),
) )
}) })
@@ -119,7 +113,14 @@ describe("pretty signature rendering", () => {
expect(pretty).toContain(" /** @deprecated */\n legacy?: string") expect(pretty).toContain(" /** @deprecated */\n legacy?: string")
expect(pretty).toContain(" /** @format uri */\n homepage?: string") expect(pretty).toContain(" /** @format uri */\n homepage?: string")
expect(pretty).toContain( expect(pretty).toContain(
[" /**", ' * @default ["a","b"]', " * @minItems 2", " * @maxItems 5", " */", " tags?: Array<string>"].join("\n"), [
" /**",
' * @default ["a","b"]',
" * @minItems 2",
" * @maxItems 5",
" */",
" tags?: Array<string>",
].join("\n"),
) )
}) })
@@ -212,7 +213,11 @@ describe("non-identifier property names render as quoted keys", () => {
const tool = Tool.make({ const tool = Tool.make({
description: "Adapter tool with awkward field names", description: "Adapter tool with awkward field names",
input: rawSchema, input: rawSchema,
output: { type: "object", properties: { "content-type": { type: "string" } }, required: ["content-type"] } as const, output: {
type: "object",
properties: { "content-type": { type: "string" } },
required: ["content-type"],
} as const,
run: () => Effect.succeed({ "content-type": "text/plain" }), run: () => Effect.succeed({ "content-type": "text/plain" }),
}) })
expect(inputTypeScript(tool)).toContain('"foo-bar"?: string') expect(inputTypeScript(tool)).toContain('"foo-bar"?: string')
@@ -269,9 +274,9 @@ describe("pretty signatures in search results", () => {
const runtime = CodeMode.make({ tools: { github: { list_issues: listIssues }, orders: { lookup: lookupOrder } } }) const runtime = CodeMode.make({ tools: { github: { list_issues: listIssues }, orders: { lookup: lookupOrder } } })
const search = async (query: string) => { const search = async (query: string) => {
const result = await Effect.runPromise(runtime.execute( const result = await Effect.runPromise(
`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`, runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`),
)) )
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) throw new Error("search failed") if (!result.ok) throw new Error("search failed")
return result.value as { items: Array<{ path: string; signature: string }>; total: number } return result.value as { items: Array<{ path: string; signature: string }>; total: number }
+111 -43
View File
@@ -40,7 +40,11 @@ describe("Date", () => {
}) })
test("UTC getters read calendar components", async () => { test("UTC getters read calendar components", async () => {
expect(await value(`const d = new Date("2024-03-05T06:07:08.009Z"); return [d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate(), d.getUTCHours(), d.getUTCMinutes(), d.getUTCSeconds(), d.getUTCMilliseconds()]`)).toEqual([2024, 2, 5, 6, 7, 8, 9]) expect(
await value(
`const d = new Date("2024-03-05T06:07:08.009Z"); return [d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate(), d.getUTCHours(), d.getUTCMinutes(), d.getUTCSeconds(), d.getUTCMilliseconds()]`,
),
).toEqual([2024, 2, 5, 6, 7, 8, 9])
}) })
test("invalid dates yield NaN times, guardable in-sandbox", async () => { test("invalid dates yield NaN times, guardable in-sandbox", async () => {
@@ -49,7 +53,9 @@ describe("Date", () => {
}) })
test("toISOString on an invalid date is a catchable error", async () => { test("toISOString on an invalid date is a catchable error", async () => {
expect(await value(`try { new Date("garbage").toISOString(); return "no" } catch { return "caught" }`)).toBe("caught") expect(await value(`try { new Date("garbage").toISOString(); return "no" } catch { return "caught" }`)).toBe(
"caught",
)
}) })
test("template interpolation renders the ISO form", async () => { test("template interpolation renders the ISO form", async () => {
@@ -72,14 +78,18 @@ describe("Date", () => {
}) })
test("sorting dates with a numeric comparator", async () => { test("sorting dates with a numeric comparator", async () => {
expect(await value(` expect(
await value(`
const dates = [new Date(3000), new Date(1000), new Date(2000)] const dates = [new Date(3000), new Date(1000), new Date(2000)]
return dates.sort((a, b) => a - b).map((d) => d.getTime()) return dates.sort((a, b) => a - b).map((d) => d.getTime())
`)).toEqual([1000, 2000, 3000]) `),
).toEqual([1000, 2000, 3000])
}) })
test("new Date(year, month, day) accepts component form", async () => { test("new Date(year, month, day) accepts component form", async () => {
expect(await value(`const d = new Date(2024, 0, 2); return [d.getFullYear(), d.getMonth(), d.getDate()]`)).toEqual([2024, 0, 2]) expect(await value(`const d = new Date(2024, 0, 2); return [d.getFullYear(), d.getMonth(), d.getDate()]`)).toEqual([
2024, 0, 2,
])
}) })
test("typeof and unknown properties are forgiving", async () => { test("typeof and unknown properties are forgiving", async () => {
@@ -95,25 +105,31 @@ describe("RegExp", () => {
}) })
test("exec exposes captures and index", async () => { test("exec exposes captures and index", async () => {
expect(await value(`const m = /a(b+)/.exec("xxabbc"); return { full: m[0], group: m[1], index: m.index }`)).toEqual({ expect(await value(`const m = /a(b+)/.exec("xxabbc"); return { full: m[0], group: m[1], index: m.index }`)).toEqual(
{
full: "abb", full: "abb",
group: "bb", group: "bb",
index: 2, index: 2,
}) },
)
expect(await value(`return /a/.exec("zzz")`)).toBeNull() expect(await value(`return /a/.exec("zzz")`)).toBeNull()
}) })
test("named groups read through", async () => { test("named groups read through", async () => {
expect(await value(`const m = /(?<word>[a-z]+)-(?<num>\\d+)/.exec("id ab-42"); return m.groups.word + m.groups.num`)).toBe("ab42") expect(
await value(`const m = /(?<word>[a-z]+)-(?<num>\\d+)/.exec("id ab-42"); return m.groups.word + m.groups.num`),
).toBe("ab42")
}) })
test("global exec advances lastIndex across calls", async () => { test("global exec advances lastIndex across calls", async () => {
expect(await value(` expect(
await value(`
const r = /\\d+/g const r = /\\d+/g
const first = r.exec("a1b22c") const first = r.exec("a1b22c")
const second = r.exec("a1b22c") const second = r.exec("a1b22c")
return [first[0], second[0]] return [first[0], second[0]]
`)).toEqual(["1", "22"]) `),
).toEqual(["1", "22"])
}) })
test("string match: non-global carries index, global lists all matches", async () => { test("string match: non-global carries index, global lists all matches", async () => {
@@ -194,34 +210,51 @@ describe("RegExp", () => {
describe("Map", () => { describe("Map", () => {
test("get/set/has/size with chaining", async () => { test("get/set/has/size with chaining", async () => {
expect(await value(` expect(
await value(`
const m = new Map() const m = new Map()
m.set("a", 1).set("b", 2) m.set("a", 1).set("b", 2)
return { a: m.get("a"), b: m.get("b"), has: m.has("a"), miss: m.get("zz") === undefined, size: m.size } return { a: m.get("a"), b: m.get("b"), has: m.has("a"), miss: m.get("zz") === undefined, size: m.size }
`)).toEqual({ a: 1, b: 2, has: true, miss: true, size: 5 - 3 }) `),
).toEqual({ a: 1, b: 2, has: true, miss: true, size: 5 - 3 })
}) })
test("object keys use identity", async () => { test("object keys use identity", async () => {
expect(await value(` expect(
await value(`
const key = { id: 1 } const key = { id: 1 }
const m = new Map() const m = new Map()
m.set(key, "hit") m.set(key, "hit")
return [m.get(key), m.get({ id: 1 }) === undefined] return [m.get(key), m.get({ id: 1 }) === undefined]
`)).toEqual(["hit", true]) `),
).toEqual(["hit", true])
}) })
test("construction from entry pairs and another Map", async () => { test("construction from entry pairs and another Map", async () => {
expect(await value(`const m = new Map([["a", 1], ["b", 2]]); return m.get("b")`)).toBe(2) expect(await value(`const m = new Map([["a", 1], ["b", 2]]); return m.get("b")`)).toBe(2)
expect(await value(`const m = new Map([["a", 1]]); const n = new Map(m); n.set("b", 2); return [n.get("a"), n.get("b"), m.has("b")]`)).toEqual([1, 2, false]) expect(
await value(
`const m = new Map([["a", 1]]); const n = new Map(m); n.set("b", 2); return [n.get("a"), n.get("b"), m.has("b")]`,
),
).toEqual([1, 2, false])
expect((await error(`return new Map("nope")`)).message).toMatch(/\[key, value\] pairs/) expect((await error(`return new Map("nope")`)).message).toMatch(/\[key, value\] pairs/)
expect((await error(`return new Map(["flat"])`)).message).toMatch(/\[key, value\] pairs/) expect((await error(`return new Map(["flat"])`)).message).toMatch(/\[key, value\] pairs/)
}) })
test("keys/values/entries return arrays", async () => { test("keys/values/entries return arrays", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
return { keys: m.keys(), values: m.values(), entries: m.entries() } return { keys: m.keys(), values: m.values(), entries: m.entries() }
`)).toEqual({ keys: ["a", "b"], values: [1, 2], entries: [["a", 1], ["b", 2]] }) `),
).toEqual({
keys: ["a", "b"],
values: [1, 2],
entries: [
["a", 1],
["b", 2],
],
})
}) })
test("Object.fromEntries(map) and Array.from(map)", async () => { test("Object.fromEntries(map) and Array.from(map)", async () => {
@@ -230,13 +263,15 @@ describe("Map", () => {
}) })
test("for...of iterates [key, value] pairs with destructuring", async () => { test("for...of iterates [key, value] pairs with destructuring", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
let total = 0 let total = 0
let names = "" let names = ""
for (const [key, count] of m) { names += key; total += count } for (const [key, count] of m) { names += key; total += count }
return names + total return names + total
`)).toBe("ab3") `),
).toBe("ab3")
}) })
test("spread produces entry pairs", async () => { test("spread produces entry pairs", async () => {
@@ -244,32 +279,38 @@ describe("Map", () => {
}) })
test("forEach passes (value, key)", async () => { test("forEach passes (value, key)", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
const seen = [] const seen = []
m.forEach((count, key) => seen.push(key + count)) m.forEach((count, key) => seen.push(key + count))
return seen return seen
`)).toEqual(["a1", "b2"]) `),
).toEqual(["a1", "b2"])
}) })
test("delete and clear", async () => { test("delete and clear", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
const removed = m.delete("a") const removed = m.delete("a")
const missed = m.delete("zz") const missed = m.delete("zz")
const sizeAfterDelete = m.size const sizeAfterDelete = m.size
m.clear() m.clear()
return [removed, missed, sizeAfterDelete, m.size] return [removed, missed, sizeAfterDelete, m.size]
`)).toEqual([true, false, 1, 0]) `),
).toEqual([true, false, 1, 0])
}) })
test("counting idiom: grouped tallies", async () => { test("counting idiom: grouped tallies", async () => {
expect(await value(` expect(
await value(`
const words = ["a", "b", "a", "c", "a"] const words = ["a", "b", "a", "c", "a"]
const counts = new Map() const counts = new Map()
for (const word of words) counts.set(word, (counts.get(word) ?? 0) + 1) for (const word of words) counts.set(word, (counts.get(word) ?? 0) + 1)
return Object.fromEntries(counts) return Object.fromEntries(counts)
`)).toEqual({ a: 3, b: 1, c: 1 }) `),
).toEqual({ a: 3, b: 1, c: 1 })
}) })
test("maps serialize to {} at the boundary, like JSON", async () => { test("maps serialize to {} at the boundary, like JSON", async () => {
@@ -286,12 +327,14 @@ describe("Map", () => {
describe("Set", () => { describe("Set", () => {
test("add/has/delete/size with chaining", async () => { test("add/has/delete/size with chaining", async () => {
expect(await value(` expect(
await value(`
const s = new Set() const s = new Set()
s.add(1).add(2).add(1) s.add(1).add(2).add(1)
const removed = s.delete(2) const removed = s.delete(2)
return [s.size, s.has(1), s.has(2), removed] return [s.size, s.has(1), s.has(2), removed]
`)).toEqual([1, true, false, true]) `),
).toEqual([1, true, false, true])
}) })
test("dedupe idiom: [...new Set(items)]", async () => { test("dedupe idiom: [...new Set(items)]", async () => {
@@ -308,11 +351,13 @@ describe("Set", () => {
}) })
test("for...of iterates values", async () => { test("for...of iterates values", async () => {
expect(await value(` expect(
await value(`
let total = 0 let total = 0
for (const n of new Set([1, 2, 3])) total += n for (const n of new Set([1, 2, 3])) total += n
return total return total
`)).toBe(6) `),
).toBe(6)
}) })
test("sets serialize to {} at the boundary, like JSON", async () => { test("sets serialize to {} at the boundary, like JSON", async () => {
@@ -338,21 +383,32 @@ describe("stdlib integration", () => {
}) })
test("dates inside Map values survive in-sandbox reads", async () => { test("dates inside Map values survive in-sandbox reads", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["start", new Date(1000)]]) const m = new Map([["start", new Date(1000)]])
return m.get("start").getTime() return m.get("start").getTime()
`)).toBe(1000) `),
).toBe(1000)
}) })
test("instanceof recognizes the stdlib value types", async () => { test("instanceof recognizes the stdlib value types", async () => {
expect(await value(`return [new Date(0) instanceof Date, /a/ instanceof RegExp, new Map() instanceof Map, new Set() instanceof Set]`)).toEqual([true, true, true, true]) expect(
expect(await value(`return [[1] instanceof Array, [1] instanceof Object, ({}) instanceof Object, 5 instanceof Object]`)).toEqual([true, true, true, false]) await value(
`return [new Date(0) instanceof Date, /a/ instanceof RegExp, new Map() instanceof Map, new Set() instanceof Set]`,
),
).toEqual([true, true, true, true])
expect(
await value(`return [[1] instanceof Array, [1] instanceof Object, ({}) instanceof Object, 5 instanceof Object]`),
).toEqual([true, true, true, false])
expect(await value(`return [new Map() instanceof Set, "s" instanceof Date]`)).toEqual([false, false]) expect(await value(`return [new Map() instanceof Set, "s" instanceof Date]`)).toEqual([false, false])
expect(await value(`const p = Promise.resolve(1); const isPromise = p instanceof Promise; await p; return isPromise`)).toBe(true) expect(
await value(`const p = Promise.resolve(1); const isPromise = p instanceof Promise; await p; return isPromise`),
).toBe(true)
}) })
test("realistic pipeline: parse, extract with regex, dedupe, count by day", async () => { test("realistic pipeline: parse, extract with regex, dedupe, count by day", async () => {
expect(await value(` expect(
await value(`
const raw = '[{"at":"2024-01-01T05:00:00Z","tag":"a b"},{"at":"2024-01-01T09:00:00Z","tag":"b c"},{"at":"2024-01-02T01:00:00Z","tag":"a"}]' const raw = '[{"at":"2024-01-01T05:00:00Z","tag":"a b"},{"at":"2024-01-01T09:00:00Z","tag":"b c"},{"at":"2024-01-02T01:00:00Z","tag":"a"}]'
const rows = JSON.parse(raw) const rows = JSON.parse(raw)
const tags = new Set() const tags = new Set()
@@ -363,27 +419,34 @@ describe("stdlib integration", () => {
byDay.set(day, (byDay.get(day) ?? 0) + 1) byDay.set(day, (byDay.get(day) ?? 0) + 1)
} }
return { tags: [...tags].sort((a, b) => (a < b ? -1 : 1)), byDay: Object.fromEntries(byDay) } return { tags: [...tags].sort((a, b) => (a < b ? -1 : 1)), byDay: Object.fromEntries(byDay) }
`)).toEqual({ tags: ["a", "b", "c"], byDay: { "2024-01-01": 2, "2024-01-02": 1 } }) `),
).toEqual({ tags: ["a", "b", "c"], byDay: { "2024-01-01": 2, "2024-01-02": 1 } })
}) })
}) })
describe("sandbox values at intra-sandbox checkpoints", () => { describe("sandbox values at intra-sandbox checkpoints", () => {
test("Object.values/entries keep Dates usable", async () => { test("Object.values/entries keep Dates usable", async () => {
expect(await value(`return Object.values({ d: new Date(0) })[0].getTime()`)).toBe(0) expect(await value(`return Object.values({ d: new Date(0) })[0].getTime()`)).toBe(0)
expect(await value(`const [key, d] = Object.entries({ d: new Date(0) })[0]; return key + ":" + d.getTime()`)).toBe("d:0") expect(await value(`const [key, d] = Object.entries({ d: new Date(0) })[0]; return key + ":" + d.getTime()`)).toBe(
"d:0",
)
}) })
test("Object.assign keeps Maps usable", async () => { test("Object.assign keeps Maps usable", async () => {
expect(await value(`const merged = Object.assign({}, { m: new Map([["a", 1]]) }); return merged.m.get("a")`)).toBe(1) expect(await value(`const merged = Object.assign({}, { m: new Map([["a", 1]]) }); return merged.m.get("a")`)).toBe(
1,
)
}) })
test("object and array spread keep sandbox values usable", async () => { test("object and array spread keep sandbox values usable", async () => {
expect(await value(` expect(
await value(`
const src = { m: new Map([["a", 1]]) } const src = { m: new Map([["a", 1]]) }
const copy = { ...src } const copy = { ...src }
copy.m.set("b", 2) copy.m.set("b", 2)
return [copy.m.get("a"), src.m.get("b")] return [copy.m.get("a"), src.m.get("b")]
`)).toEqual([1, 2]) `),
).toEqual([1, 2])
expect(await value(`const list = [new Date(1000)]; const copy = [...list]; return copy[0].getTime()`)).toBe(1000) expect(await value(`const list = [new Date(1000)]; const copy = [...list]; return copy[0].getTime()`)).toBe(1000)
}) })
@@ -404,7 +467,10 @@ describe("sandbox values at intra-sandbox checkpoints", () => {
}) })
test("the host boundary still serializes JSON forms: results, JSON.stringify, and tool arguments", async () => { test("the host boundary still serializes JSON forms: results, JSON.stringify, and tool arguments", async () => {
expect(await value(`return { d: new Date(0), m: new Map([["a", 1]]) }`)).toEqual({ d: "1970-01-01T00:00:00.000Z", m: {} }) expect(await value(`return { d: new Date(0), m: new Map([["a", 1]]) }`)).toEqual({
d: "1970-01-01T00:00:00.000Z",
m: {},
})
expect(await value(`return JSON.stringify({ d: new Date(0) })`)).toBe('{"d":"1970-01-01T00:00:00.000Z"}') expect(await value(`return JSON.stringify({ d: new Date(0) })`)).toBe('{"d":"1970-01-01T00:00:00.000Z"}')
const observed: Array<unknown> = [] const observed: Array<unknown> = []
@@ -417,10 +483,12 @@ describe("sandbox values at intra-sandbox checkpoints", () => {
return "ok" return "ok"
}), }),
}) })
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
CodeMode.execute({
tools: { host: { capture } }, tools: { host: { capture } },
code: `return await tools.host.capture({ when: new Date(0), tags: new Map([["a", 1]]) })`, code: `return await tools.host.capture({ when: new Date(0), tags: new Map([["a", 1]]) })`,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(observed).toStrictEqual([{ when: "1970-01-01T00:00:00.000Z", tags: {} }]) expect(observed).toStrictEqual([{ when: "1970-01-01T00:00:00.000Z", tags: {} }])
}) })
+1 -4
View File
@@ -206,10 +206,7 @@ export const prepare = Effect.fn("LLMRequestPrep.prepare")(function* (input: Pre
}) })
function resolveTools(input: Pick<PrepareInput, "tools" | "agent" | "permission" | "user">) { function resolveTools(input: Pick<PrepareInput, "tools" | "agent" | "permission" | "user">) {
const visible = Permission.visibleTools( const visible = Permission.visibleTools(input.tools, Permission.merge(input.agent.permission, input.permission ?? []))
input.tools,
Permission.merge(input.agent.permission, input.permission ?? []),
)
return Record.filter(visible, (_, k) => input.user.tools?.[k] !== false) return Record.filter(visible, (_, k) => input.user.tools?.[k] !== false)
} }
+7 -3
View File
@@ -104,7 +104,8 @@ export function groupByServer(
const byLongest = [...servers].sort((a, b) => b.length - a.length) const byLongest = [...servers].sort((a, b) => b.length - a.length)
const groups = new Map<string, CatalogEntry[]>() const groups = new Map<string, CatalogEntry[]>()
for (const key of Object.keys(mcpTools).sort((a, b) => a.localeCompare(b))) { for (const key of Object.keys(mcpTools).sort((a, b) => a.localeCompare(b))) {
const server = byLongest.find((name) => key.startsWith(name + "_")) ?? (key.includes("_") ? key.slice(0, key.indexOf("_")) : key) const server =
byLongest.find((name) => key.startsWith(name + "_")) ?? (key.includes("_") ? key.slice(0, key.indexOf("_")) : key)
const local = server && key.startsWith(server + "_") ? key.slice(server.length + 1) : key const local = server && key.startsWith(server + "_") ? key.slice(server.length + 1) : key
const def = mcpDefs[key] const def = mcpDefs[key]
const entry: CatalogEntry = { const entry: CatalogEntry = {
@@ -129,7 +130,9 @@ export function buildCatalog(
mcpDefs: Record<string, MCPToolDef>, mcpDefs: Record<string, MCPToolDef>,
servers: readonly string[], servers: readonly string[],
): CatalogEntry[] { ): CatalogEntry[] {
return [...groupByServer(mcpTools, servers, mcpDefs).values()].flat().filter((entry) => entry.tool.execute !== undefined) return [...groupByServer(mcpTools, servers, mcpDefs).values()]
.flat()
.filter((entry) => entry.tool.execute !== undefined)
} }
/** /**
@@ -334,7 +337,8 @@ export const CodeModeTool = Tool.define(
const collect = (attachment: Attachment) => void attachments.push(attachment) const collect = (attachment: Attachment) => void attachments.push(attachment)
// Stream the current call list to the UI. Sent on every status change so the // Stream the current call list to the UI. Sent on every status change so the
// tool part shows each child call appearing and resolving while the program runs. // tool part shows each child call appearing and resolving while the program runs.
const publish = () => ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } }) const publish = () =>
ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } })
// One CodeMode tool per MCP tool, running the same shared middle as legacy // One CodeMode tool per MCP tool, running the same shared middle as legacy
// per-tool registration (McpInvoke.invoke: plugin before hook → permission // per-tool registration (McpInvoke.invoke: plugin before hook → permission
+4 -2
View File
@@ -276,7 +276,10 @@ const layer = Layer.effect(
// fresh per turn so it tracks live tool-list changes. Hard-denied tools (the shared // fresh per turn so it tracks live tool-list changes. Hard-denied tools (the shared
// Permission.visibleTools predicate over the agent's ruleset) never enter the // Permission.visibleTools predicate over the agent's ruleset) never enter the
// catalog, its inlined signatures, or the in-program search index. // catalog, its inlined signatures, or the in-program search index.
const describeCodeMode = Effect.fn("ToolRegistry.describeCodeMode")(function* (agent: Agent.Info, permission?: PermissionV1.Ruleset) { const describeCodeMode = Effect.fn("ToolRegistry.describeCodeMode")(function* (
agent: Agent.Info,
permission?: PermissionV1.Ruleset,
) {
const visible = Permission.visibleTools(yield* mcp.tools(), Permission.merge(agent.permission, permission ?? [])) const visible = Permission.visibleTools(yield* mcp.tools(), Permission.merge(agent.permission, permission ?? []))
const servers = Object.keys(yield* mcp.clients()).map(McpCatalog.sanitize) const servers = Object.keys(yield* mcp.clients()).map(McpCatalog.sanitize)
return catalogInstructions(visible, yield* mcp.defs(), servers) return catalogInstructions(visible, yield* mcp.defs(), servers)
@@ -341,7 +344,6 @@ const layer = Layer.effect(
}), }),
) )
function isZodType(value: unknown): value is z.ZodType { function isZodType(value: unknown): value is z.ZodType {
return typeof value === "object" && value !== null && "_zod" in value return typeof value === "object" && value !== null && "_zod" in value
} }
+1 -2
View File
@@ -96,8 +96,7 @@ function resolveTools(trigger?: Plugin.Interface["trigger"]) {
Layer.mergeAll( Layer.mergeAll(
Layer.mock(Permission.Service, { ask: () => Effect.void }), Layer.mock(Permission.Service, { ask: () => Effect.void }),
Layer.mock(Plugin.Service, { Layer.mock(Plugin.Service, {
trigger: trigger: trigger ?? (((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"]),
trigger ?? (((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"]),
}), }),
Layer.mock(Truncate.Service, { Layer.mock(Truncate.Service, {
output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), output: (text: string) => Effect.succeed({ content: text, truncated: false as const }),
@@ -21,8 +21,7 @@ import type { Tool as AITool } from "ai"
import { Effect, Layer } from "effect" import { Effect, Layer } from "effect"
// A 1x1 transparent PNG, base64-encoded, used to exercise image attachments. // A 1x1 transparent PNG, base64-encoded, used to exercise image attachments.
const PNG = const PNG = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
const SERVER = "fixtures" const SERVER = "fixtures"
@@ -163,8 +162,8 @@ async function buildTool() {
// this real in-memory server listed — the same snapshot shape the live service returns. // this real in-memory server listed — the same snapshot shape the live service returns.
const layer = Layer.mergeAll( const layer = Layer.mergeAll(
Layer.mock(Plugin.Service, { Layer.mock(Plugin.Service, {
trigger: (((_name: unknown, _input: unknown, output: unknown) => trigger: ((_name: unknown, _input: unknown, output: unknown) =>
Effect.succeed(output)) as Plugin.Interface["trigger"]), Effect.succeed(output)) as Plugin.Interface["trigger"],
}), }),
Layer.mock(Truncate.Service, { Layer.mock(Truncate.Service, {
output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), output: (text: string) => Effect.succeed({ content: text, truncated: false as const }),
@@ -196,9 +195,7 @@ describe("code mode integration (real MCP server)", () => {
test("the appended catalog inlines full signatures with real MCP schemas", () => { test("the appended catalog inlines full signatures with real MCP schemas", () => {
expect(description).toContain("Available tools (COMPLETE list") expect(description).toContain("Available tools (COMPLETE list")
expect(description).toContain("- fixtures (4 tools)") expect(description).toContain("- fixtures (4 tools)")
expect(description).toContain( expect(description).toContain("tools.fixtures.add(input: { a: number; b: number }): Promise<{ sum: number }>")
"tools.fixtures.add(input: { a: number; b: number }): Promise<{ sum: number }>",
)
expect(description).toContain("tools.fixtures.get_text(input: { name: string }): Promise<unknown>") expect(description).toContain("tools.fixtures.get_text(input: { name: string }): Promise<unknown>")
expect(description).toContain("// Add two numbers and return the structured sum") expect(description).toContain("// Add two numbers and return the structured sum")
// Small catalog: everything is inline, so no discovery tool is advertised. // Small catalog: everything is inline, so no discovery tool is advertised.
+22 -16
View File
@@ -193,7 +193,9 @@ describe("code mode execute", () => {
// never cherry-picks a catalog tool or fabricates result fields. // never cherry-picks a catalog tool or fabricates result fields.
expect(description).toContain("## Workflow") expect(description).toContain("## Workflow")
expect(description).toContain("1. Pick a tool from the list under `## Available tools`") expect(description).toContain("1. Pick a tool from the list under `## Available tools`")
expect(description).toContain('`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string') expect(description).toContain(
'`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string',
)
expect(description).toContain("Return only the fields you need") expect(description).toContain("Return only the fields you need")
expect(description).not.toContain("total_count") expect(description).not.toContain("total_count")
}) })
@@ -249,7 +251,9 @@ describe("code mode execute", () => {
expect(description).toContain("tools.$codemode.search(") expect(description).toContain("tools.$codemode.search(")
// PARTIAL catalogs put search first in the workflow and advertise namespace browsing. // PARTIAL catalogs put search first in the workflow and advertise namespace browsing.
expect(description).toContain("1. Find a tool (skip when it is already listed below)") expect(description).toContain("1. Find a tool (skip when it is already listed below)")
expect(description).toContain('- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.') expect(description).toContain(
'- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.',
)
expect(description).not.toContain("total_count") expect(description).not.toContain("total_count")
// All op lines cost the same estimated tokens (chars/4 rounds away the 1- vs 3-digit // All op lines cost the same estimated tokens (chars/4 rounds away the 1- vs 3-digit
// name difference), so the path tiebreak decides: the lexicographically-first ops made // name difference), so the path tiebreak decides: the lexicographically-first ops made
@@ -290,7 +294,10 @@ describe("code mode execute", () => {
linear_search: mcpTool("search", () => ""), linear_search: mcpTool("search", () => ""),
}) })
const output = await Effect.runPromise( const output = await Effect.runPromise(
tool.execute({ code: "const namespaces = Object.keys(tools); return { namespaces, count: namespaces.length }" }, ctx), tool.execute(
{ code: "const namespaces = Object.keys(tools); return { namespaces, count: namespaces.length }" },
ctx,
),
) )
expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear"], count: 2 }) expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear"], count: 2 })
}) })
@@ -565,9 +572,7 @@ describe("code mode execute", () => {
const tool = await build({ const tool = await build({
shot_take: mcpTool("take", () => ({ content: [{ type: "image", data: "PNGDATA", mimeType: "image/png" }] })), shot_take: mcpTool("take", () => ({ content: [{ type: "image", data: "PNGDATA", mimeType: "image/png" }] })),
}) })
const out = await Effect.runPromise( const out = await Effect.runPromise(tool.execute({ code: "await tools.shot.take({}); return 'captured'" }, ctx))
tool.execute({ code: "await tools.shot.take({}); return 'captured'" }, ctx),
)
expect(out.output).toBe("captured") expect(out.output).toBe("captured")
expect(out.attachments).toHaveLength(1) expect(out.attachments).toHaveLength(1)
}) })
@@ -690,9 +695,7 @@ describe("code mode permission visibility", () => {
expect(called).toEqual([]) expect(called).toEqual([])
// The rest of the namespace still works. // The rest of the namespace still works.
const allowed = await Effect.runPromise( const allowed = await Effect.runPromise(tool.execute({ code: "return await tools.github.list_issues({})" }, ctx))
tool.execute({ code: "return await tools.github.list_issues({})" }, ctx),
)
expect(allowed.metadata.error).toBeUndefined() expect(allowed.metadata.error).toBeUndefined()
expect(allowed.output).toBe("ok") expect(allowed.output).toBe("ok")
}) })
@@ -706,9 +709,7 @@ describe("code mode permission visibility", () => {
["github"], ["github"],
[askRule("github_list_issues")], [askRule("github_list_issues")],
) )
const out = await Effect.runPromise( const out = await Effect.runPromise(tool.execute({ code: "return await tools.github.list_issues({})" }, askCtx))
tool.execute({ code: "return await tools.github.list_issues({})" }, askCtx),
)
expect(out.output).toBe("ok") expect(out.output).toBe("ok")
expect(asked).toEqual(["github_list_issues"]) expect(asked).toEqual(["github_list_issues"])
}) })
@@ -733,16 +734,21 @@ describe("toSandboxResult", () => {
test("prefers structuredContent over text", () => { test("prefers structuredContent over text", () => {
const { collect } = collector() const { collect } = collector()
expect(toSandboxResult({ structuredContent: { x: 1 }, content: [{ type: "text", text: "hi" }] }, collect)).toEqual( expect(toSandboxResult({ structuredContent: { x: 1 }, content: [{ type: "text", text: "hi" }] }, collect)).toEqual({
{ x: 1 }, x: 1,
) })
}) })
test("joins text content when no structured content is present", () => { test("joins text content when no structured content is present", () => {
const { collect } = collector() const { collect } = collector()
expect( expect(
toSandboxResult( toSandboxResult(
{ content: [{ type: "text", text: "one" }, { type: "text", text: "two" }] }, {
content: [
{ type: "text", text: "one" },
{ type: "text", text: "two" },
],
},
collect, collect,
), ),
).toBe("one\ntwo") ).toBe("one\ntwo")
+7 -1
View File
@@ -2355,7 +2355,13 @@ function Execute(props: ToolProps) {
return ( return (
<> <>
<InlineTool <InlineTool
icon={props.part.state.status === "completed" && !hasRuntimeError() ? "✓" : props.part.state.status === "error" || hasRuntimeError() ? "✗" : "│"} icon={
props.part.state.status === "completed" && !hasRuntimeError()
? "✓"
: props.part.state.status === "error" || hasRuntimeError()
? "✗"
: "│"
}
color={hasRuntimeError() ? theme.error : undefined} color={hasRuntimeError() ? theme.error : undefined}
spinner={isLoading()} spinner={isLoading()}
pending="execute" pending="execute"