chore: generate

This commit is contained in:
opencode-agent[bot] 2026-07-03 04:49:44 +00:00
commit 83c638eaac
20 changed files with 2137 additions and 1207 deletions

View file

@ -8,11 +8,11 @@ The package is currently private to this workspace. Its API is designed around t
```ts ```ts
// One execution // One execution
yield* CodeMode.execute({ tools, code }) yield * CodeMode.execute({ tools, code })
// A reusable runtime // A reusable runtime
const runtime = CodeMode.make({ tools, limits }) const runtime = CodeMode.make({ tools, limits })
yield* runtime.execute(code) yield * runtime.execute(code)
// One agent-facing code tool // One agent-facing code tool
const codeTool = runtime.agentTool() const codeTool = runtime.agentTool()
@ -55,7 +55,9 @@ const runtime = CodeMode.make({
}, },
}) })
const result = yield* runtime.execute(` const result =
yield *
runtime.execute(`
const order = await tools.orders.lookup({ id: "order_42" }) const order = await tools.orders.lookup({ id: "order_42" })
return { id: order.id, needsAttention: order.status !== "complete" } return { id: order.id, needsAttention: order.status !== "complete" }
`) `)
@ -72,8 +74,8 @@ Successful result values are JSON-safe data. A program that returns `undefined`,
```ts ```ts
const tool = Tool.make({ const tool = Tool.make({
description, description,
input, // Effect Schema (validating) or JSON Schema (render-only) input, // Effect Schema (validating) or JSON Schema (render-only)
output, // optional; same choice output, // optional; same choice
run, run,
}) })
``` ```
@ -89,13 +91,15 @@ The description and schemas are part of the model-visible tool contract. Keep de
Use `CodeMode.execute` for a single execution: Use `CodeMode.execute` for a single execution:
```ts ```ts
const result = yield* CodeMode.execute({ const result =
tools: { orders: { lookup: lookupOrder } }, yield *
code: `return await tools.orders.lookup({ id: "order_42" })`, CodeMode.execute({
limits: { maxToolCalls: 10 }, tools: { orders: { lookup: lookupOrder } },
onToolCallStart: (call) => Effect.logDebug("CodeMode tool started", call), code: `return await tools.orders.lookup({ id: "order_42" })`,
onToolCallEnd: (call) => Effect.logDebug("CodeMode tool settled", call), limits: { maxToolCalls: 10 },
}) onToolCallStart: (call) => Effect.logDebug("CodeMode tool started", call),
onToolCallEnd: (call) => Effect.logDebug("CodeMode tool settled", call),
})
``` ```
The Effect environment is inferred from the supplied tools. CodeMode does not erase service requirements introduced by tool implementations. The Effect environment is inferred from the supplied tools. CodeMode does not erase service requirements introduced by tool implementations.
@ -110,10 +114,10 @@ const runtime = CodeMode.make({
limits: { timeoutMs: 30_000 }, limits: { timeoutMs: 30_000 },
}) })
runtime.catalog() // structured tool descriptions runtime.catalog() // structured tool descriptions
runtime.instructions() // model-facing syntax and tool guide runtime.instructions() // model-facing syntax and tool guide
runtime.execute(source) // ExecuteResult runtime.execute(source) // ExecuteResult
runtime.agentTool() // { name, description, input, output, execute } runtime.agentTool() // { name, description, input, output, execute }
``` ```
`catalog`, `instructions`, and `agentTool` are projections of the same configured tool tree. `agentTool().description` is exactly `instructions()`. `catalog`, `instructions`, and `agentTool` are projections of the same configured tool tree. `agentTool().description` is exactly `instructions()`.
@ -222,10 +226,10 @@ CodeMode is an orchestration language, not a general JavaScript runtime.
The limits are exactly three knobs: The limits are exactly three knobs:
| Limit | Default | Bounds | | Limit | Default | Bounds |
| --- | ---: | --- | | ---------------- | -------------------: | -------------------------------------------------------------------- |
| `timeoutMs` | none — no timeout | Wall-clock execution time. | | `timeoutMs` | none — no timeout | Wall-clock execution time. |
| `maxToolCalls` | none — unlimited | Tool calls admitted during the execution. | | `maxToolCalls` | none — unlimited | Tool calls admitted during the execution. |
| `maxOutputBytes` | none — no truncation | Model-facing output: the serialized result value plus captured logs. | | `maxOutputBytes` | none — no truncation | Model-facing output: the serialized result value plus captured logs. |
No limit has a default, on purpose: execution budgets are host policy, not library policy — a host that wants a bound sets one; a host that can interrupt the execution fiber (as OpenCode does on user cancel) may set no timeout, and a host with its own tool-output truncation (as OpenCode has) may leave `maxOutputBytes` unset. A host with neither should set `maxOutputBytes`, or oversized results silently flood model context. No limit has a default, on purpose: execution budgets are host policy, not library policy — a host that wants a bound sets one; a host that can interrupt the execution fiber (as OpenCode does on user cancel) may set no timeout, and a host with its own tool-output truncation (as OpenCode has) may leave `maxOutputBytes` unset. A host with neither should set `maxOutputBytes`, or oversized results silently flood model context.
@ -254,28 +258,25 @@ Two interpreter internals are fixed constants rather than knobs: at most 8 tool
Failures are data: Failures are data:
| Kind | Meaning | | Kind | Meaning |
| --- | --- | | ----------------------- | -------------------------------------------------------------------------------------------------------- |
| `ParseError` | Source is empty or cannot be parsed. | | `ParseError` | Source is empty or cannot be parsed. |
| `UnsupportedSyntax` | Parsed JavaScript is outside the supported subset. | | `UnsupportedSyntax` | Parsed JavaScript is outside the supported subset. |
| `UnknownTool` | A program referenced a tool the host did not provide. | | `UnknownTool` | A program referenced a tool the host did not provide. |
| `InvalidToolInput` | Tool input failed schema decoding or safe-data copying. | | `InvalidToolInput` | Tool input failed schema decoding or safe-data copying. |
| `InvalidToolOutput` | Tool output failed schema decoding or safe-data copying. | | `InvalidToolOutput` | Tool output failed schema decoding or safe-data copying. |
| `InvalidDataValue` | Program data violated the plain-data contract (depth, circularity, blocked properties, non-data values). | | `InvalidDataValue` | Program data violated the plain-data contract (depth, circularity, blocked properties, non-data values). |
| `ToolCallLimitExceeded` | Calls exceeded `maxToolCalls`. | | `ToolCallLimitExceeded` | Calls exceeded `maxToolCalls`. |
| `TimeoutExceeded` | Execution exceeded `timeoutMs`. | | `TimeoutExceeded` | Execution exceeded `timeoutMs`. |
| `ToolFailure` | A tool refused or failed. | | `ToolFailure` | A tool refused or failed. |
| `ExecutionFailure` | The program threw or another execution error occurred. | | `ExecutionFailure` | The program threw or another execution error occurred. |
Unknown host failures, defects, invalid outputs, and copying failures are sanitized. To return a safe operational refusal, fail with `toolError`: Unknown host failures, defects, invalid outputs, and copying failures are sanitized. To return a safe operational refusal, fail with `toolError`:
```ts ```ts
import { toolError } from "@opencode-ai/codemode" import { toolError } from "@opencode-ai/codemode"
run: ({ id }) => run: ({ id }) => (authorized(id) ? loadOrder(id) : Effect.fail(toolError("Order is unavailable")))
authorized(id)
? loadOrder(id)
: Effect.fail(toolError("Order is unavailable"))
``` ```
Only the supplied message is model-visible. The optional cause is never returned in `ExecuteResult`; hosts should perform any required internal logging before crossing this boundary. Only the supplied message is model-visible. The optional cause is never returned in `ExecuteResult`; hosts should perform any required internal logging before crossing this boundary.

View file

@ -39,6 +39,7 @@ package and was **deleted** in Wave 3 (done, see below).
From issue #34787 and design discussion. Do not relitigate these casually. From issue #34787 and design discussion. Do not relitigate these casually.
### Core direction ### Core direction
- Generic CodeMode lives in its own package: `@opencode-ai/codemode` (repo scope convention; - Generic CodeMode lives in its own package: `@opencode-ai/codemode` (repo scope convention;
the issue's `@opencode/codemode` name was normalized to the `@opencode-ai/*` convention). the issue's `@opencode/codemode` name was normalized to the `@opencode-ai/*` convention).
- **Keep the hand-rolled interpreter.** No QuickJS/V8/sandbox-engine dependency. We own and - **Keep the hand-rolled interpreter.** No QuickJS/V8/sandbox-engine dependency. We own and
@ -52,6 +53,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
products/blog posts) in code, comments, commit messages, or docs in this repo. products/blog posts) in code, comments, commit messages, or docs in this repo.
### MCP / tools ### MCP / tools
- The MCP adapter lives in OpenCode, not here. It converts MCP definitions into ordinary - The MCP adapter lives in OpenCode, not here. It converts MCP definitions into ordinary
`Tool.make(...)` definitions and hands CodeMode a plain tool tree. `Tool.make(...)` definitions and hands CodeMode a plain tool tree.
- Permissions stay in the OpenCode adapter (each tool's `run` wraps the permission ask). - Permissions stay in the OpenCode adapter (each tool's `run` wraps the permission ask).
@ -61,8 +63,9 @@ From issue #34787 and design discussion. Do not relitigate these casually.
`tools.<server>.<tool>` namespaces before handing them over. `tools.<server>.<tool>` namespaces before handing them over.
### Discovery / search ### Discovery / search
- **Search only — no separate `describe`.** `tools.$codemode.search({ query?, namespace?, - **Search only — no separate `describe`.** `tools.$codemode.search({ query?, namespace?,
limit? })` over the final tool tree, owned by this package. limit? })` over the final tool tree, owned by this package.
- Search result item shape: `{ path, description, signature }` in an `{ items, total }` - Search result item shape: `{ path, description, signature }` in an `{ items, total }`
wrapper. The `signature` string embeds the full input/output TypeScript types — in search wrapper. The `signature` string embeds the full input/output TypeScript types — in search
results it is the pretty, JSDoc-annotated multiline form (Fix 7), so per-field schema results it is the pretty, JSDoc-annotated multiline form (Fix 7), so per-field schema
@ -82,6 +85,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
- Tools without an output schema render `unknown` as their return type. - Tools without an output schema render `unknown` as their return type.
### Schemas / Tool.make ### Schemas / Tool.make
- `Tool.make` carries rich metadata so search can render real signatures. - `Tool.make` carries rich metadata so search can render real signatures.
- Support **Effect Schema** (first-class, validating) and **JSON Schema** (initially - Support **Effect Schema** (first-class, validating) and **JSON Schema** (initially
render-only — used for TypeScript rendering; the adapter may validate on its own). Leave render-only — used for TypeScript rendering; the adapter may validate on its own). Leave
@ -90,6 +94,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
normalization for plugin authors can come later. normalization for plugin authors can come later.
### Attachments / output ### Attachments / output
- **No `output.text/file/image` API in v1.** (Deleted in Wave 2.) - **No `output.text/file/image` API in v1.** (Deleted in Wave 2.)
- Tool calls return native structured payloads into the sandbox. Files/images emitted by - Tool calls return native structured payloads into the sandbox. Files/images emitted by
child tools **never enter the sandbox** — the OpenCode adapter strips and accumulates them child tools **never enter the sandbox** — the OpenCode adapter strips and accumulates them
@ -100,6 +105,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
image bytes into context or drop attachments. image bytes into context or drop attachments.
### Runtime behavior ### Runtime behavior
- Limits are EXACTLY the three public knobs: `{ timeoutMs, maxToolCalls, maxOutputBytes }` - Limits are EXACTLY the three public knobs: `{ timeoutMs, maxToolCalls, maxOutputBytes }`
matching the original locked spec exactly. NO limit has a default (user direction, Fix 6 matching the original locked spec exactly. NO limit has a default (user direction, Fix 6
for the first two; extended to `maxOutputBytes` in the truncation-layering fix below): for the first two; extended to `maxOutputBytes` in the truncation-layering fix below):
@ -129,7 +135,7 @@ From issue #34787 and design discussion. Do not relitigate these casually.
- `console.*` is captured into `logs` on the result; the host appends them to model-facing - `console.*` is captured into `logs` on the result; the host appends them to model-facing
output. Not a tool call; costs no tool budget. output. Not a tool call; costs no tool budget.
- Simple tool-call **start/end hooks** for nested progress: `onToolCallStart({ index, name, - Simple tool-call **start/end hooks** for nested progress: `onToolCallStart({ index, name,
input })` and `onToolCallEnd({ index, name, input, durationMs, outcome, message? })`. input })` and `onToolCallEnd({ index, name, input, durationMs, outcome, message? })`.
Interrupted calls fire no end event. No `CurrentToolCall` context service (removed in Interrupted calls fire no end event. No `CurrentToolCall` context service (removed in
Wave 2). Wave 2).
@ -148,8 +154,9 @@ and `bun run typecheck`; from `packages/opencode`, `bun run typecheck` and
`test/tool/registry.test.ts`). `test/tool/registry.test.ts`).
### Wave 0 — scaffold (done) ### Wave 0 — scaffold (done)
- `packages/codemode` created from the experiments implementation: `src/{index,codemode,tool, - `packages/codemode` created from the experiments implementation: `src/{index,codemode,tool,
tool-error,tool-runtime}.ts`, README, AGENTS.md, tests. tool-error,tool-runtime}.ts`, README, AGENTS.md, tests.
- `package.json`: name `@opencode-ai/codemode`, deps `acorn@8.15.0`, `typescript: catalog:`, - `package.json`: name `@opencode-ai/codemode`, deps `acorn@8.15.0`, `typescript: catalog:`,
`effect: catalog:` (both repos pin effect `4.0.0-beta.83`; opencode's effect patch only `effect: catalog:` (both repos pin effect `4.0.0-beta.83`; opencode's effect patch only
touches `unstable/httpapi`, which this package doesn't use). touches `unstable/httpapi`, which this package doesn't use).
@ -157,6 +164,7 @@ and `bun run typecheck`; from `packages/opencode`, `bun run typecheck` and
Context.Service key string renamed to `@opencode-ai/codemode/CurrentToolCall`. Context.Service key string renamed to `@opencode-ai/codemode/CurrentToolCall`.
### Wave 1a — forgiving JS semantics (done) ### Wave 1a — forgiving JS semantics (done)
Ported from the old opencode rune work; `test/parity.test.ts` (24 tests) is the acceptance Ported from the old opencode rune work; `test/parity.test.ts` (24 tests) is the acceptance
spec. The seeded interpreter was deliberately strict; these behaviors replaced that: spec. The seeded interpreter was deliberately strict; these behaviors replaced that:
@ -173,6 +181,7 @@ spec. The seeded interpreter was deliberately strict; these behaviors replaced t
null/undefined still throws (real JS throws too). null/undefined still throws (real JS throws too).
### Wave 1b-i — stdlib value types: Date, RegExp, Map, Set (done) ### Wave 1b-i — stdlib value types: Date, RegExp, Map, Set (done)
`src/values.ts` holds `SandboxDate/SandboxRegExp/SandboxMap/SandboxSet` (own module so both `src/values.ts` holds `SandboxDate/SandboxRegExp/SandboxMap/SandboxSet` (own module so both
`codemode.ts` and `tool-runtime.ts` import without a cycle). Design: `codemode.ts` and `tool-runtime.ts` import without a cycle). Design:
@ -202,19 +211,20 @@ spec. The seeded interpreter was deliberately strict; these behaviors replaced t
interpolation renders `/regex/` and ISO dates directly. interpolation renders `/regex/` and ISO dates directly.
### Wave 2 — API layer (done) ### Wave 2 — API layer (done)
The package's public contract, reshaped for the Wave 3 adapter. 101 tests / 0 fail after this The package's public contract, reshaped for the Wave 3 adapter. 101 tests / 0 fail after this
wave; both packages typecheck clean. wave; both packages typecheck clean.
- **`Tool.make` schema flexibility** (`src/tool.ts`): `input`/`output` each accept an Effect - **`Tool.make` schema flexibility** (`src/tool.ts`): `input`/`output` each accept an Effect
Schema (validating, decoded both directions as before) OR a raw JSON Schema document Schema (validating, decoded both directions as before) OR a raw JSON Schema document
(render-only — no validation, values pass through; rendering handles `$defs`/`definitions` (render-only — no validation, values pass through; rendering handles `$defs`/`definitions`
+ `$ref`). `output` is **optional** → signature renders `Promise<unknown>` and the host - `$ref`). `output` is **optional** → signature renders `Promise<unknown>` and the host
result is exposed as-is. Discrimination via `Schema.isSchema`. New helpers exported from result is exposed as-is. Discrimination via `Schema.isSchema`. New helpers exported from
`tool.ts`: `inputTypeScript`/`outputTypeScript`/`decodeInput`/`decodeOutput`/ `tool.ts`: `inputTypeScript`/`outputTypeScript`/`decodeInput`/`decodeOutput`/
`jsonSchemaToTypeScript`; `tool-runtime.ts` consumes them (no direct `Schema.*` use there `jsonSchemaToTypeScript`; `tool-runtime.ts` consumes them (no direct `Schema.*` use there
anymore). Types `JsonSchema`/`ToolSchema` exported from the index. Note: an empty anymore). Types `JsonSchema`/`ToolSchema` exported from the index. Note: an empty
`Schema.Struct({})` renders as `{ } | Array<unknown>` (effect's JSON Schema emission) — `Schema.Struct({})` renders as `{ } | Array<unknown>` (effect's JSON Schema emission) —
cosmetic, fixed in Wave 4. cosmetic, fixed in Wave 4.
- **`output.*` API deleted**: `OutputItem`(+Schema), result `output` fields, the `output` - **`output.*` API deleted**: `OutputItem`(+Schema), result `output` fields, the `output`
global/namespace dispatch, `invokeOutput`/`outputItem`/helpers, interpreter output fields, global/namespace dispatch, `invokeOutput`/`outputItem`/helpers, interpreter output fields,
instructions line, README section, seeded tests. AGENTS.md keeps a rephrased instructions line, README section, seeded tests. AGENTS.md keeps a rephrased
@ -228,22 +238,23 @@ wave; both packages typecheck clean.
(`ToolError`/`ToolRuntimeError` message, else "Tool execution failed"). Interrupted calls (`ToolError`/`ToolRuntimeError` message, else "Tool execution failed"). Interrupted calls
fire no end event (timeout kills the whole execution anyway). fire no end event (timeout kills the whole execution anyway).
- **Limits collapse**: public `ExecutionLimits` = `{ timeoutMs?, maxToolCalls?, - **Limits collapse**: public `ExecutionLimits` = `{ timeoutMs?, maxToolCalls?,
maxOutputBytes? }` (defaults 10_000 / 100 / 32_000). This wave kept the other knobs as maxOutputBytes? }` (defaults 10_000 / 100 / 32_000). This wave kept the other knobs as
internal defaults reachable through an `@internal` `InternalExecutionLimits` type; Fix 5 internal defaults reachable through an `@internal` `InternalExecutionLimits` type; Fix 5
later deleted that type and the internal limit system entirely. later deleted that type and the internal limit system entirely.
- **`maxOutputBytes` truncation** (CodeMode-owned, never fails): applied via `boundOutput` in - **`maxOutputBytes` truncation** (CodeMode-owned, never fails): applied via `boundOutput` in
a final `Effect.map` over every result path (success/timeout/normalized failure). Oversized a final `Effect.map` over every result path (success/timeout/normalized failure). Oversized
serialized values become truncated text + ` [result truncated: N bytes exceeds the M-byte serialized values become truncated text + ` [result truncated: N bytes exceeds the M-byte
output limit; return a smaller value]`; logs keep leading lines within the remaining budget output limit; return a smaller value]`; logs keep leading lines within the remaining budget
+ `[logs truncated: showing K of N lines]`; result gains `truncated: true` (also added to - `[logs truncated: showing K of N lines]`; result gains `truncated: true` (also added to
`ExecuteResultSchema`). UTF-8-safe truncation (no split code points). (The in-sandbox `ExecuteResultSchema`). UTF-8-safe truncation (no split code points). (The in-sandbox
`maxDataBytes` check that used to throw first on oversized raw values died in Fix 5 — `maxDataBytes` check that used to throw first on oversized raw values died in Fix 5 —
truncation is now the only result-size mechanism.) truncation is now the only result-size mechanism.)
- **Search polish**: default limit 12 → **10** (`defaultSearchLimit`); exact-path lookup — a - **Search polish**: default limit 12 → **10** (`defaultSearchLimit`); exact-path lookup — a
trimmed query equal to one tool path (optionally `tools.`-prefixed) returns that tool alone trimmed query equal to one tool path (optionally `tools.`-prefixed) returns that tool alone
(`total: 1`), bypassing ranking. Tokenization/ranking/shape unchanged. (`total: 1`), bypassing ranking. Tokenization/ranking/shape unchanged.
### Wave 3 — OpenCode MCP adapter (done) ### Wave 3 — OpenCode MCP adapter (done)
`packages/opencode/src/session/code-mode.ts` rewritten as a thin adapter over this package; `packages/opencode/src/session/code-mode.ts` rewritten as a thin adapter over this package;
the vendored rune interpreter is gone. Same `define(mcpTools, mcpDefs, servers)` signature, so the vendored rune interpreter is gone. Same `define(mcpTools, mcpDefs, servers)` signature, so
`tools.ts` gating (flag on + MCP tools exist → single `execute` tool, early-return suppresses `tools.ts` gating (flag on + MCP tools exist → single `execute` tool, early-return suppresses
@ -257,7 +268,7 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
invoked) — so signature rendering, the inline-vs-search switch, and `$codemode.search` invoked) — so signature rendering, the inline-vs-search switch, and `$codemode.search`
availability all come from this package and stay consistent with execution. availability all come from this package and stay consistent with execution.
- **`run` path**: per-child permission ask first (`ctx.ask({ permission: entry.key, patterns: - **`run` path**: per-child permission ask first (`ctx.ask({ permission: entry.key, patterns:
["*"], always: ["*"] })`, exactly the old gating; approving `execute` approves no child). ["*"], always: ["*"] })`, exactly the old gating; approving `execute` approves no child).
Denials and host failures are mapped to `toolError(message)` so they surface as safe, Denials and host failures are mapped to `toolError(message)` so they surface as safe,
catchable in-program failures (MCP `isError` text propagates as `e.message`; without this catchable in-program failures (MCP `isError` text propagates as `e.message`; without this
they'd be sanitized to "Tool execution failed"). Dispatch reuses the ai-sdk wrapper from they'd be sanitized to "Tool execution failed"). Dispatch reuses the ai-sdk wrapper from
@ -270,10 +281,10 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
No handles, no `Result<T>` envelope, no base64 in the sandbox, no data-size tuning (the No handles, no `Result<T>` envelope, no base64 in the sandbox, no data-size tuning (the
`maxDataBytes` budget that existed at the time was deleted in Fix 5). `maxDataBytes` budget that existed at the time was deleted in Fix 5).
- **Execute result**: `{ output: formatValue(value) + trailing "Logs:" section (success AND - **Execute result**: `{ output: formatValue(value) + trailing "Logs:" section (success AND
error — logs are plain pre-formatted lines now), attachments: accumulated }` through the error — logs are plain pre-formatted lines now), attachments: accumulated }` through the
existing `Tool.ExecuteResult.attachments``message-v2.ts` vision plumbing; attachments existing `Tool.ExecuteResult.attachments``message-v2.ts` vision plumbing; attachments
ride on both success and error results. Diagnostic `suggestions` not already contained in ride on both success and error results. Diagnostic `suggestions` not already contained in
the message are appended to error output. Native outer truncation stays on (adapter never the message are appended to error output. Native outer truncation stays on (adapter never
sets `metadata.truncated`); CodeMode's own `maxOutputBytes` (32 KB default at the time) sets `metadata.truncated`); CodeMode's own `maxOutputBytes` (32 KB default at the time)
cut first — since the truncation-layering fix, native truncation is the only layer. cut first — since the truncation-layering fix, native truncation is the only layer.
Limits: `{ timeoutMs: 30_000 }` at the time (matched the default MCP request timeout); Limits: `{ timeoutMs: 30_000 }` at the time (matched the default MCP request timeout);
@ -296,6 +307,7 @@ per-MCP registration; MCP resource tools unaffected) is unchanged.
describe/`renderType`/`rankTools` tests died with the old design (58+17+24 → 34+16). describe/`renderType`/`rankTools` tests died with the old design (58+17+24 → 34+16).
### Wave 4 — instructions/prompting + polish (done) ### Wave 4 — instructions/prompting + polish (done)
Instructions are now the budgeted-catalog + prompting-guidance form; verified e2e against a Instructions are now the budgeted-catalog + prompting-guidance form; verified e2e against a
real MCP config. Package still 101 tests / 0 fail; opencode adapter suites still 34 + 16; both real MCP config. Package still 101 tests / 0 fail; opencode adapter suites still 34 + 16; both
packages typecheck clean. packages typecheck clean.
@ -321,7 +333,7 @@ packages typecheck clean.
exported); `CodeMode.execute` (one-shot) passes it too, preserving the exported); `CodeMode.execute` (one-shot) passes it too, preserving the
`execute``make().execute` law. A speculative `tools.$codemode.search` call on a small `execute``make().execute` law. A speculative `tools.$codemode.search` call on a small
catalog now succeeds instead of `UnknownTool`, and unknown-tool suggestions always point at catalog now succeeds instead of `UnknownTool`, and unknown-tool suggestions always point at
search. Search is *advertised* in the instructions only when the inlined list is PARTIAL, search. Search is _advertised_ in the instructions only when the inlined list is PARTIAL,
keeping small-catalog instructions tight. keeping small-catalog instructions tight.
- **Prompting content** in `instructions()`, mapping 1:1 to the §5 transcript failures: - **Prompting content** in `instructions()`, mapping 1:1 to the §5 transcript failures:
parse-string-results-as-JSON, return-small, console-for-intermediates, and parse-string-results-as-JSON, return-small, console-for-intermediates, and
@ -340,7 +352,7 @@ packages typecheck clean.
- **E2E (verified, headless)**: from the repo root with `OPENCODE_EXPERIMENTAL_CODE_MODE=1`, - **E2E (verified, headless)**: from the repo root with `OPENCODE_EXPERIMENTAL_CODE_MODE=1`,
the scratch `.opencode/opencode.jsonc` (context7, github, playwright, sentry, memory, the scratch `.opencode/opencode.jsonc` (context7, github, playwright, sentry, memory,
sequential-thinking; left uncommitted/as-is), and `bun packages/opencode/src/index.ts run sequential-thinking; left uncommitted/as-is), and `bun packages/opencode/src/index.ts run
--dangerously-skip-permissions -m opencode/claude-sonnet-4-5 "..."`. Confirmed: a single --dangerously-skip-permissions -m opencode/claude-sonnet-4-5 "..."`. Confirmed: a single
`execute` tool registered alongside core tools (per-MCP registration suppressed; MCP `execute` tool registered alongside core tools (per-MCP registration suppressed; MCP
resource tools unaffected); the live description read back as "Available tools (PARTIAL — resource tools unaffected); the live description read back as "Available tools (PARTIAL —
56 of 88 shown; find the rest with tools.$codemode.search):" with correct per-namespace 56 of 88 shown; find the rest with tools.$codemode.search):" with correct per-namespace
@ -352,6 +364,7 @@ packages typecheck clean.
images, output truncation. images, output truncation.
### Wave 5 — Promise generalization (done) ### Wave 5 — Promise generalization (done)
First-class promise values in the interpreter; the direct-tool-call-only `Promise.all` First-class promise values in the interpreter; the direct-tool-call-only `Promise.all`
restriction (and its bespoke AST checks) is gone. Package suite is 136 tests / 0 fail (35 new restriction (and its bespoke AST checks) is gone. Package suite is 136 tests / 0 fail (35 new
in `test/promise.test.ts`); adapter suites and both typechecks unchanged/green; the opencode in `test/promise.test.ts`); adapter suites and both typechecks unchanged/green; the opencode
@ -495,7 +508,7 @@ adapter needed **no changes**.
`tools.*`." (the second line drops the tools clause when the tree is empty). `tools.*`." (the second line drops the tools clause when the tree is empty).
- **`## Workflow`**: numbered steps — find a tool via `tools.$codemode.search` → read - **`## Workflow`**: numbered steps — find a tool via `tools.$codemode.search` → read
the `{ path, description, signature }` matches → call by path → `typeof res === the `{ path, description, signature }` matches → call by path → `typeof res ===
"string" ? JSON.parse(res) : res` → return only the needed fields. When the catalog is "string" ? JSON.parse(res) : res` → return only the needed fields. When the catalog is
COMPLETE the search/read steps collapse into "Pick a tool from the list under COMPLETE the search/read steps collapse into "Pick a tool from the list under
`## Available tools`" and the steps renumber (4 instead of 5). `## Available tools`" and the steps renumber (4 instead of 5).
- **`## Rules`**: call-by-exact-path; TEXT-is-JSON → JSON.parse; return small (never raw - **`## Rules`**: call-by-exact-path; TEXT-is-JSON → JSON.parse; return small (never raw
@ -526,70 +539,72 @@ adapter needed **no changes**.
**Fix 4 — token-budgeted catalog (was bytes)** (user direction: signatures need a token **Fix 4 — token-budgeted catalog (was bytes)** (user direction: signatures need a token
budget; namespaces must always be present): budget; namespaces must always be present):
- `src/token.ts` added: copy of `@opencode-ai/core/util/token` (`round(chars / 4)`), so
the package stays dependency-free; keep in sync if the core heuristic changes. - `src/token.ts` added: copy of `@opencode-ai/core/util/token` (`round(chars / 4)`), so
- `DiscoveryOptions.maxInlineCatalogBytes``maxInlineCatalogTokens` (default 4,000 the package stays dependency-free; keep in sync if the core heuristic changes.
estimated tokens ≈ the old 16,000 bytes at 4 chars/token — behavior parity, not a size - `DiscoveryOptions.maxInlineCatalogBytes``maxInlineCatalogTokens` (default 4,000
reduction). `discoveryPlan` charges `estimate(catalogLine(tool))` per line; cheapest-first estimated tokens ≈ the old 16,000 bytes at 4 chars/token — behavior parity, not a size
+ stop-on-first-miss unchanged at the time (stop-on-first-miss replaced by round-robin in reduction). `discoveryPlan` charges `estimate(catalogLine(tool))` per line; cheapest-first
- stop-on-first-miss unchanged at the time (stop-on-first-miss replaced by round-robin in
Fix 8). Namespace stub lines were and remain unbudgeted — every Fix 8). Namespace stub lines were and remain unbudgeted — every
namespace always appears with its tool count, even at budget 0 (asserted in package and namespace always appears with its tool count, even at budget 0 (asserted in package and
adapter tests). adapter tests).
- Ripple: chars/4 rounding erases small line-length differences, so equal-cost lines fall - Ripple: chars/4 rounding erases small line-length differences, so equal-cost lines fall
to the lexicographic path tiebreak; the adapter's PARTIAL test now asserts the to the lexicographic path tiebreak; the adapter's PARTIAL test now asserts the
lexicographic tail (`op_99`) is excluded instead of `op_149`. Fixed-prose measurements lexicographic tail (`op_99`) is excluded instead of `op_149`. Fixed-prose measurements
(2026-07): preamble ~44 + Workflow ~146 + Rules ~362 + Syntax ~453 ≈ 1,100 tokens fixed; (2026-07): preamble ~44 + Workflow ~146 + Rules ~362 + Syntax ~453 ≈ 1,100 tokens fixed;
worst-case net description ≈ fixed + 4,000 ≈ 5,100 estimated tokens. worst-case net description ≈ fixed + 4,000 ≈ 5,100 estimated tokens.
**Fix 5 — internal limits removed** (user direction: only the three PUBLIC limits survive as **Fix 5 — internal limits removed** (user direction: only the three PUBLIC limits survive as
configurable knobs; the internal limit system dies): configurable knobs; the internal limit system dies):
- `ExecutionLimits` (`timeoutMs` 10_000 / `maxToolCalls` 100 / `maxOutputBytes` 32_000 at
the time; Fix 6 later removed the first two defaults. Same validation: safe integers, - `ExecutionLimits` (`timeoutMs` 10_000 / `maxToolCalls` 100 / `maxOutputBytes` 32_000 at
timeoutMs >= 1, others >= 0, RangeError otherwise) is now the time; Fix 6 later removed the first two defaults. Same validation: safe integers,
the ENTIRE limit surface — exactly the shape §2's original locked spec named. timeoutMs >= 1, others >= 0, RangeError otherwise) is now
`ResolvedExecutionLimits` shrank to those three fields; the `@internal` the ENTIRE limit surface — exactly the shape §2's original locked spec named.
`InternalExecutionLimits` type is deleted. `ResolvedExecutionLimits` shrank to those three fields; the `@internal`
- **Deleted outright**: `maxOperations` and the whole operation-budget machinery `InternalExecutionLimits` type is deleted.
(`recordWork`/`recordOperation`/`budget.operations`, plus the `workUnits`/ - **Deleted outright**: `maxOperations` and the whole operation-budget machinery
`cheapArrayMethods` cost helpers); `maxSourceBytes` (the pre-parse source-size check); (`recordWork`/`recordOperation`/`budget.operations`, plus the `workUnits`/
`maxDataBytes` (every byte-accounting path: `runtimeValueBytes`, `boundedProgramValue`, `cheapArrayMethods` cost helpers); `maxSourceBytes` (the pre-parse source-size check);
the container-size caches (`containerSizes`/`objectCounts`), Map/Set incremental `bytes` `maxDataBytes` (every byte-accounting path: `runtimeValueBytes`, `boundedProgramValue`,
fields in `values.ts`, string-growth `limitString` checks, tool-argument/result byte the container-size caches (`containerSizes`/`objectCounts`), Map/Set incremental `bytes`
checks in `tool-runtime.ts`, and the final-result size check); `maxAuditBytes` (log and fields in `values.ts`, string-growth `limitString` checks, tool-argument/result byte
audit-trail byte accounting — `toolCalls` records and the start/end hooks are unchanged); checks in `tool-runtime.ts`, and the final-result size check); `maxAuditBytes` (log and
`maxCollectionLength` (every array-length/object-field-count check — this knob was audit-trail byte accounting — `toolCalls` records and the start/end hooks are unchanged);
actively harmful: an MCP tool returning 20k rows failed). The `OperationLimitExceeded` `maxCollectionLength` (every array-length/object-field-count check — this knob was
and `AuditLimitExceeded` diagnostic kinds are gone from the `DiagnosticKind` union and actively harmful: an MCP tool returning 20k rows failed). The `OperationLimitExceeded`
`ExecuteResultSchema` (fine — the package is unreleased). and `AuditLimitExceeded` diagnostic kinds are gone from the `DiagnosticKind` union and
- **Fixed constants, not knobs**: `TOOL_CALL_CONCURRENCY = 8` (codemode.ts; the fork `ExecuteResultSchema` (fine — the package is unreleased).
semaphore) and `MAX_VALUE_DEPTH = 32` (tool-runtime.ts; the `copyIn` depth check — kept - **Fixed constants, not knobs**: `TOOL_CALL_CONCURRENCY = 8` (codemode.ts; the fork
only because it produces a clearer error than a native stack-overflow RangeError; still semaphore) and `MAX_VALUE_DEPTH = 32` (tool-runtime.ts; the `copyIn` depth check — kept
`InvalidDataValue`). The `DataLimits` plumbing through `tool-runtime.ts` is gone — only because it produces a clearer error than a native stack-overflow RangeError; still
`copyIn(value, label)` needs no limits argument, and `ToolRuntime.make` takes just `InvalidDataValue`). The `DataLimits` plumbing through `tool-runtime.ts` is gone —
`(tools, maxToolCalls, hooks?, searchIndex?)`. `copyIn(value, label)` needs no limits argument, and `ToolRuntime.make` takes just
- **Verified fact**: timeout interruption does NOT depend on the operation budget — the `(tools, maxToolCalls, hooks?, searchIndex?)`.
Effect fiber runtime auto-yields between interpreter steps, so `timeoutMs` interrupts - **Verified fact**: timeout interruption does NOT depend on the operation budget — the
even a pure `while (true) {}` loop (empirically verified: a 200ms timeout fired at Effect fiber runtime auto-yields between interpreter steps, so `timeoutMs` interrupts
~225ms with maxOperations set to MAX_SAFE_INTEGER before the deletion). A regression even a pure `while (true) {}` loop (empirically verified: a 200ms timeout fired at
test in `codemode.test.ts` asserts exactly this (`while(true){}` + `timeoutMs: 200` ~225ms with maxOperations set to MAX_SAFE_INTEGER before the deletion). A regression
`TimeoutExceeded`, elapsed well under a few seconds). test in `codemode.test.ts` asserts exactly this (`while(true){}` + `timeoutMs: 200`
- **Kept (correctness, not budgets)**: circular detection (`copyIn` walks + `TimeoutExceeded`, elapsed well under a few seconds).
`rejectCircularInsertion` on mutations), plain-objects-only, blocked properties - **Kept (correctness, not budgets)**: circular detection (`copyIn` walks +
(`__proto__`/`constructor`/`prototype`), data-only checks, and all three public-limit `rejectCircularInsertion` on mutations), plain-objects-only, blocked properties
behaviors unchanged. (`__proto__`/`constructor`/`prototype`), data-only checks, and all three public-limit
- Behavior deltas beyond the intended kills: in-sandbox structures deeper than 32 levels behaviors unchanged.
now fail at the data boundary (`copyIn`) instead of at construction; array index - Behavior deltas beyond the intended kills: in-sandbox structures deeper than 32 levels
assignment allows any non-negative integer index (holes permitted, message now "must be now fail at the data boundary (`copyIn`) instead of at construction; array index
a non-negative integer"); interpreter-produced deep/hostile structures that overflow the assignment allows any non-negative integer index (holes permitted, message now "must be
native stack during a walk still normalize to the existing "Execution exceeded the a non-negative integer"); interpreter-produced deep/hostile structures that overflow the
maximum nesting depth." data diagnostic — failures remain data everywhere. native stack during a walk still normalize to the existing "Execution exceeded the
- Tests: deleted the knob-only tests (stdlib Map/Set collection-length growth ×2, maximum nesting depth." data diagnostic — failures remain data everywhere.
enumeration operation-budget, codemode maxDataBytes/maxSourceBytes/maxOperations/ - Tests: deleted the knob-only tests (stdlib Map/Set collection-length growth ×2,
maxConcurrency-RangeError assertions, and the adapter's runaway-loop-via-operation-limit enumeration operation-budget, codemode maxDataBytes/maxSourceBytes/maxOperations/
test — superseded by the package timeout regression test); rewrote the helpers that used maxConcurrency-RangeError assertions, and the adapter's runaway-loop-via-operation-limit
`InternalExecutionLimits` as a convenience to plain `ExecutionLimits` test — superseded by the package timeout regression test); rewrote the helpers that used
(promise/enumeration/stdlib run helpers). Package suite: 154 pass / 0 fail; adapter `InternalExecutionLimits` as a convenience to plain `ExecutionLimits`
suites: 34 + 16. (promise/enumeration/stdlib run helpers). Package suite: 154 pass / 0 fail; adapter
suites: 34 + 16.
**Fix 6 — no default timeout / tool-call cap** (user direction): `timeoutMs` and **Fix 6 — no default timeout / tool-call cap** (user direction): `timeoutMs` and
`maxToolCalls` lost their defaults (were 10_000 / 100) — absent now means no timeout / `maxToolCalls` lost their defaults (were 10_000 / 100) — absent now means no timeout /
@ -639,196 +654,200 @@ adapter suites: 34 + 16.
**Fix 8 — condensed instructions + round-robin catalog fairness + plural-aware search** **Fix 8 — condensed instructions + round-robin catalog fairness + plural-aware search**
(user direction: the fixed instruction prose was too verbose; two discovery fixes ride (user direction: the fixed instruction prose was too verbose; two discovery fixes ride
along). All in `tool-runtime.ts`; no interpreter changes. along). All in `tool-runtime.ts`; no interpreter changes.
- **Syntax section inverted**: the three dense allowlist lines (~453 estimated tokens)
are replaced by four short lines (~188) built on "models already know JavaScript; name - **Syntax section inverted**: the three dense allowlist lines (~453 estimated tokens)
only what is unusual or missing": (1) standard modern JS works — functions/closures, are replaced by four short lines (~188) built on "models already know JavaScript; name
destructuring, template literals, loops, try/catch, spread, optional chaining, the only what is unusual or missing": (1) standard modern JS works — functions/closures,
usual Array/String/Object/Math/JSON methods, plus Date/RegExp/Map/Set and destructuring, template literals, loops, try/catch, spread, optional chaining, the
Promise.all/allSettled/race/resolve/reject; (2) TypeScript type annotations are usual Array/String/Object/Math/JSON methods, plus Date/RegExp/Map/Set and
stripped before execution, decorators are not supported; (3) NOT supported (each fails Promise.all/allSettled/race/resolve/reject; (2) TypeScript type annotations are
with a message naming the alternative): classes, generators, for await...of, stripped before execution, decorators are not supported; (3) NOT supported (each fails
.then/.catch/.finally (use await with try/catch), `x instanceof Error` (caught errors with a message naming the alternative): classes, generators, for await...of,
are plain `{ name, message }` objects), splice; (4) the data-boundary note (Dates → .then/.catch/.finally (use await with try/catch), `x instanceof Error` (caught errors
ISO strings; Map/Set/RegExp → `{}`). Every claim was verified against the interpreter are plain `{ name, message }` objects), splice; (4) the data-boundary note (Dates →
before writing: probed empirically — classes/generators/for-await/.then/.catch/ ISO strings; Map/Set/RegExp → `{}`). Every claim was verified against the interpreter
.finally/`instanceof Error`/splice/decorators/BigInt/labeled statements/tagged before writing: probed empirically — classes/generators/for-await/.then/.catch/
templates/object getters all fail with clear diagnostics; TS annotations/`as`/ .finally/`instanceof Error`/splice/decorators/BigInt/labeled statements/tagged
interfaces/type aliases are stripped and TS **enums actually work** (transpileModule templates/object getters all fail with clear diagnostics; TS annotations/`as`/
compiles them to an IIFE the interpreter runs), hence enums deliberately unmentioned. interfaces/type aliases are stripped and TS **enums actually work** (transpileModule
`supportedSyntaxMessage` (the in-diagnostic text in `codemode.ts`) is untouched. compiles them to an IIFE the interpreter runs), hence enums deliberately unmentioned.
- **Workflow/Rules deduped**: the call-by-exact-path, JSON.parse-string-results, and `supportedSyntaxMessage` (the in-diagnostic text in `codemode.ts`) is untouched.
return-small content now lives ONLY in the numbered Workflow steps (with their - **Workflow/Rules deduped**: the call-by-exact-path, JSON.parse-string-results, and
compliance-driving justifications inline: "most tools return JSON as a string", "raw return-small content now lives ONLY in the numbered Workflow steps (with their
payloads get truncated and waste context"); Rules keeps only bullets adding new compliance-driving justifications inline: "most tools return JSON as a string", "raw
content — filter/aggregate collections in code, console.* intermediates (logs ride payloads get truncated and waste context"); Rules keeps only bullets adding new
back), Promise.all parallelism, Object.keys/for...in enumeration, browse-namespace content — filter/aggregate collections in code, console.\* intermediates (logs ride
(PARTIAL only), and the media rule compressed to one line. The no-.then/.catch back), Promise.all parallelism, Object.keys/for...in enumeration, browse-namespace
guidance moved to the Syntax not-supported line. Content upgrades: the PARTIAL search (PARTIAL only), and the media rule compressed to one line. The no-.then/.catch
step gained query-style guidance (`— short phrases like "list issues" work best`; a guidance moved to the Syntax not-supported line. Content upgrades: the PARTIAL search
clearly-a-query-string example, not a tool name), and the exact-path guidance is now step gained query-style guidance (`— short phrases like "list issues" work best`; a
"call it with the result's `path` as-is (never guess segments)" / COMPLETE: "use it clearly-a-query-string example, not a tool name), and the exact-path guidance is now
as-is rather than guessing segments". "call it with the result's `path` as-is (never guess segments)" / COMPLETE: "use it
- **Fixed-prose measurements** (instructions split on `"\n## "`, catalog budget 0, as-is rather than guessing segments".
bytes/3.7 — same method as Fix 4; chars/4 in parentheses): - **Fixed-prose measurements** (instructions split on `"\n## "`, catalog budget 0,
preamble 44 → 44 (41 → 41), Workflow 146 → 187 (135 → 171), Rules 362 → 191 bytes/3.7 — same method as Fix 4; chars/4 in parentheses):
(332 → 176), Syntax 453 → 188 (419 → 174); fixed prose total 1,005 → 610 (927 → 562), preamble 44 → 44 (41 → 41), Workflow 146 → 187 (135 → 171), Rules 362 → 191
≈ 40% reduction with no behavioral content dropped. Workflow grew slightly because it (332 → 176), Syntax 453 → 188 (419 → 174); fixed prose total 1,005 → 610 (927 → 562),
absorbed the deduped parse/return-small justifications. ≈ 40% reduction with no behavioral content dropped. Workflow grew slightly because it
- **Round-robin namespace inlining** (`discoveryPlan`): the ported stop-on-first-miss absorbed the deduped parse/return-small justifications.
behavior (alphabetically-late namespaces starved to "none shown" while an early - **Round-robin namespace inlining** (`discoveryPlan`): the ported stop-on-first-miss
namespace inlines everything) is replaced by round-robin fairness — in each round behavior (alphabetically-late namespaces starved to "none shown" while an early
(namespaces alphabetical), every namespace still holding un-inlined tools attempts to namespace inlines everything) is replaced by round-robin fairness — in each round
place its next-cheapest line against the shared token budget; a namespace whose next (namespaces alphabetical), every namespace still holding un-inlined tools attempts to
line does not fit is done while the others keep going; stop when all are done. Every place its next-cheapest line against the shared token budget; a namespace whose next
namespace gets some representation before any namespace gets everything. Kept: line does not fit is done while the others keep going; stop when all are done. Every
`estimate` (chars/4) budget accounting, unbudgeted namespace stub lines, per-namespace namespace gets some representation before any namespace gets everything. Kept:
`(N tools)`/`(N tools, K shown)`/`(N tools, none shown)` labels, COMPLETE vs PARTIAL `estimate` (chars/4) budget accounting, unbudgeted namespace stub lines, per-namespace
header, alphabetical namespace order in the output, cheapest-first within each `(N tools)`/`(N tools, K shown)`/`(N tools, none shown)` labels, COMPLETE vs PARTIAL
namespace's shown set. header, alphabetical namespace order in the output, cheapest-first within each
- **Plural/singular search fix**: `tokenize`d terms matched one-directionally (term must namespace's shown set.
be substring of indexed text), so query "issues" missed a tool whose text only says - **Plural/singular search fix**: `tokenize`d terms matched one-directionally (term must
"issue". Now each term expands to `termForms` — the term plus naive singular variants be substring of indexed text), so query "issues" missed a tool whose text only says
(trailing "es" stripped when length > 3, trailing "s" when length > 2) — and each of "issue". Now each term expands to `termForms` — the term plus naive singular variants
the four field checks passes when ANY form matches. Weights, exact-path lookup, and (trailing "es" stripped when length > 3, trailing "s" when length > 2) — and each of
namespace scoping untouched. A true plural path match still outranks a singular-only the four field checks passes when ANY form matches. Weights, exact-path lookup, and
description match (path substring 8 + searchable 2 > description 4 + searchable 2). namespace scoping untouched. A true plural path match still outranks a singular-only
- **Tests**: package instruction/structure assertions updated to the new text; new description match (path substring 8 + searchable 2 > description 4 + searchable 2).
syntax-section test (leads with "Standard modern JavaScript works", names the - **Tests**: package instruction/structure assertions updated to the new text; new
verified not-supported list, keeps the data-boundary note); the budget-exhaustion syntax-section test (leads with "Standard modern JavaScript works", names the
test rewritten to assert the new fairness (alpha.expensive not fitting must NOT verified not-supported list, keeps the data-boundary note); the budget-exhaustion
prevent beta.cheap from showing: PARTIAL 2 of 3, `- beta (1 tool)` fully shown); new test rewritten to assert the new fairness (alpha.expensive not fitting must NOT
plural/singular test (query "issues" finds a singular-only tool; ranking still prevent beta.cheap from showing: PARTIAL 2 of 3, `- beta (1 tool)` fully shown); new
prefers the true "issues" path match). Adapter: description assertions updated; the plural/singular test (query "issues" finds a singular-only tool; ranking still
large-catalog PARTIAL test now asserts `zeta_only_tool` IS shown (`- zeta (1 tool)` + prefers the true "issues" path match). Adapter: description assertions updated; the
its inlined line) — it was "none shown" under starvation. README updated (budgeted large-catalog PARTIAL test now asserts `zeta_only_tool` IS shown (`- zeta (1 tool)` +
catalog paragraph → round-robin; search paragraph → singular variants; its inlined line) — it was "none shown" under starvation. README updated (budgeted
instructions-structure paragraph → new section contents). Package suite: 169 pass / catalog paragraph → round-robin; search paragraph → singular variants;
0 fail; adapter suites: 34 + 16. instructions-structure paragraph → new section contents). Package suite: 169 pass /
0 fail; adapter suites: 34 + 16.
**Fix 9 — prompting trims per user review of Fix 8** (user reviewed the condensed **Fix 9 — prompting trims per user review of Fix 8** (user reviewed the condensed
instructions and directed further cuts): instructions and directed further cuts):
- Default `maxInlineCatalogTokens` 4,000 → **2,000** (user wants ~2k tokens of signatures
auto-inlined; round-robin fairness from Fix 8 spreads it across all namespaces). - Default `maxInlineCatalogTokens` 4,000 → **2,000** (user wants ~2k tokens of signatures
- Console rule and files/images rule DROPPED from `## Rules`. Replaced by a single auto-inlined; round-robin fairness from Fix 8 spreads it across all namespaces).
`unknown`-treatment warning: "A result typed `Promise<unknown>` has no guaranteed - Console rule and files/images rule DROPPED from `## Rules`. Replaced by a single
shape — verify what actually came back before relying on its fields." (Deliberately `unknown`-treatment warning: "A result typed `Promise<unknown>` has no guaranteed
does NOT suggest console.log — user review: naming it there nudges models to log AND shape — verify what actually came back before relying on its fields." (Deliberately
return the same data; the prompt stays console-neutral, neither for nor against.) does NOT suggest console.log — user review: naming it there nudges models to log AND
The media-stripping MECHANISM is unchanged and still tested; only the prose about it return the same data; the prompt stays console-neutral, neither for nor against.)
is gone — the `[N images attached]` marker is self-explanatory in context. The media-stripping MECHANISM is unchanged and still tested; only the prose about it
- Kept as-is per user: the JSON.parse workflow step (maps to the original motivating is gone — the `[N images attached]` marker is self-explanatory in context.
transcript failure; NOT copied from prior art — see §5 note), the browse-namespace rule - Kept as-is per user: the JSON.parse workflow step (maps to the original motivating
(undecided), no no-fetch/ambient-authority rule added (proposed, not approved). transcript failure; NOT copied from prior art — see §5 note), the browse-namespace rule
- Explicitly REJECTED for now: auto-parsing JSON-looking text results at the adapter (undecided), no no-fetch/ambient-authority rule added (proposed, not approved).
boundary ("could get weird" — type flips, program-sees vs tool-sent divergence). Logged - Explicitly REJECTED for now: auto-parsing JSON-looking text results at the adapter
as a next-iteration follow-up below. boundary ("could get weird" — type flips, program-sees vs tool-sent divergence). Logged
as a next-iteration follow-up below.
**DSL-expansion pass — interpreter-surface batch from §4** (the deferred medium-tier JS **DSL-expansion pass — interpreter-surface batch from §4** (the deferred medium-tier JS
parity items, done as one focused pass; no public API or limit changes): parity items, done as one focused pass; no public API or limit changes):
- **`instanceof` + real Error values**: the `errorConstructors` names (`Error`,
`TypeError`, `RangeError`, `SyntaxError`, `ReferenceError`, `EvalError`, `URIError`) are - **`instanceof` + real Error values**: the `errorConstructors` names (`Error`,
bound globals (`ErrorConstructorReference`, callable with or without `new`; `typeof` `TypeError`, `RangeError`, `SyntaxError`, `ReferenceError`, `EvalError`, `URIError`) are
`"function"`). Error values stay the same plain `{ name, message }` null-prototype bound globals (`ErrorConstructorReference`, callable with or without `new`; `typeof`
objects as before — the constructor name additionally rides on a NON-ENUMERABLE symbol `"function"`). Error values stay the same plain `{ name, message }` null-prototype
key (`ErrorBrand`), which every `Object.entries`-based walk (copyIn/copyOut, spread, objects as before — the constructor name additionally rides on a NON-ENUMERABLE symbol
JSON.stringify) is blind to, so serialization is byte-identical to the old shape and the key (`ErrorBrand`), which every `Object.entries`-based walk (copyIn/copyOut, spread,
brand is lost on spread/boundary copies exactly like JS loses the prototype. JSON.stringify) is blind to, so serialization is byte-identical to the old shape and the
`caughtErrorValue` produces `{ name, message }` wrappers via `createErrorValue`, so brand is lost on spread/boundary copies exactly like JS loses the prototype.
caught interpreter AND tool failures are `instanceof Error` and carry the `name` the `caughtErrorValue` produces `{ name, message }` wrappers via `createErrorValue`, so
equivalent real-JS failure would have (follow-up fix, user-directed — "closest to real caught interpreter AND tool failures are `instanceof Error` and carry the `name` the
JS"): `InterpreterRuntimeError` gained an `errorName` field ("Error" default) set equivalent real-JS failure would have (follow-up fix, user-directed — "closest to real
fluently at throw sites via `.as(name)``JSON.parse` failures are `"SyntaxError"` (and JS"): `InterpreterRuntimeError` gained an `errorName` field ("Error" default) set
now include the engine's position detail in the message; safe — derived from the fluently at throw sites via `.as(name)``JSON.parse` failures are `"SyntaxError"` (and
program-supplied string), invalid regex patterns/flags `"SyntaxError"`, unknown now include the engine's position detail in the message; safe — derived from the
identifiers and TDZ access `"ReferenceError"`, assignment to a constant `"TypeError"`, program-supplied string), invalid regex patterns/flags `"SyntaxError"`, unknown
a bad `normalize` form `"RangeError"`; a host Error reaching the catch path directly identifiers and TDZ access `"ReferenceError"`, assignment to a constant `"TypeError"`,
keeps its own name when it is one of the standard seven. Tool failures and everything a bad `normalize` form `"RangeError"`; a host Error reaching the catch path directly
without a specific analogue stay `"Error"` — internal class names never leak. Specific keeps its own name when it is one of the standard seven. Tool failures and everything
names satisfy the specific `instanceof` (`e instanceof SyntaxError`), matching JS. without a specific analogue stay `"Error"` — internal class names never leak. Specific
The operator is handled in `evaluateBinaryExpression` names satisfy the specific `instanceof` (`e instanceof SyntaxError`), matching JS.
BEFORE the data-only operand check (like `typeof`, it observes any lhs — promises and The operator is handled in `evaluateBinaryExpression`
functions included); recognized rhs: the error constructors (a specific type matches its BEFORE the data-only operand check (like `typeof`, it observes any lhs — promises and
own brand or `Error`, never a sibling), `Date`/`RegExp`/`Map`/`Set` (sandbox classes), functions included); recognized rhs: the error constructors (a specific type matches its
`Array`, `Object` (any object/function-ish value), `Promise` (`SandboxPromise`), and own brand or `Error`, never a sibling), `Date`/`RegExp`/`Map`/`Set` (sandbox classes),
`Number`/`String`/`Boolean` (always false — no boxed values exist); anything else is a `Array`, `Object` (any object/function-ish value), `Promise` (`SandboxPromise`), and
catchable error naming the recognized constructors. `Number`/`String`/`Boolean` (always false — no boxed values exist); anything else is a
- **Array methods**: `splice` (mutating, returns the removed elements; insertions run catchable error naming the recognized constructors.
`rejectCircularInsertion` like push/unshift; one-arg form removes to the end, undefined - **Array methods**: `splice` (mutating, returns the removed elements; insertions run
delete count removes nothing), `fill` (circular-checked value) and `copyWithin` `rejectCircularInsertion` like push/unshift; one-arg form removes to the end, undefined
(host-delegated), and `keys`/`values`/`entries` returning **arrays** (the Map/Set delete count removes nothing), `fill` (circular-checked value) and `copyWithin`
convention — for...of and spread work either way). The `retryableArrayMethods` (host-delegated), and `keys`/`values`/`entries` returning **arrays** (the Map/Set
"rewrite using map/filter" hint set emptied out and was deleted with its branch; unknown convention — for...of and spread work either way). The `retryableArrayMethods`
array properties still read `undefined`. "rewrite using map/filter" hint set emptied out and was deleted with its branch; unknown
- **String methods**: `localeCompare(that)` (locale/options arguments ignored — host array properties still read `undefined`.
default locale; the dominant use is a sort comparator), `normalize(form?)` (invalid form - **String methods**: `localeCompare(that)` (locale/options arguments ignored — host
→ catchable error naming the four valid forms), `trimLeft`/`trimRight` as default locale; the dominant use is a sort comparator), `normalize(form?)` (invalid form
trimStart/trimEnd aliases. → catchable error naming the four valid forms), `trimLeft`/`trimRight` as
- **Actionable regex failures**: `toHostRegex` and `constructRegExp` now show the trimStart/trimEnd aliases.
offending pattern (or flags) plus the engine reason (deduped "Invalid regular - **Actionable regex failures**: `toHostRegex` and `constructRegExp` now show the
expression:" prefix via `regexFailureReason`) and a shared escaping hint offending pattern (or flags) plus the engine reason (deduped "Invalid regular
(`escapeRegexHint`); flags failures list the valid flag letters; the expression:" prefix via `regexFailureReason`) and a shared escaping hint
replaceAll/matchAll missing-`g` errors spell out the exact `/pattern/g` to write and (`escapeRegexHint`); flags failures list the valid flag letters; the
the single-match alternative. replaceAll/matchAll missing-`g` errors spell out the exact `/pattern/g` to write and
- **copyIn split (the important one)**: `copyIn(value, label, preserveSandboxValues = the single-match alternative.
false)` — recursion moved to a private `copyBounded`; `boundedData` (every intra-sandbox - **copyIn split (the important one)**: `copyIn(value, label, preserveSandboxValues =
checkpoint: `Object.*` helpers, coercion/Array.from/join inputs, template false)` — recursion moved to a private `copyBounded`; `boundedData` (every intra-sandbox
interpolation, expression-result checkpoints) is now `copyIn(value, label, true)`, checkpoint: `Object.*` helpers, coercion/Array.from/join inputs, template
which passes `SandboxDate`/`SandboxRegExp`/`SandboxMap`/`SandboxSet` through **by interpolation, expression-result checkpoints) is now `copyIn(value, label, true)`,
reference as leaves** (contents not walked — Map/Set members are validated at their which passes `SandboxDate`/`SandboxRegExp`/`SandboxMap`/`SandboxSet` through **by
mutation sites) while keeping the depth (`MAX_VALUE_DEPTH`), circularity, reference as leaves** (contents not walked — Map/Set members are validated at their
plain-objects-only, blocked-property, and data-only checks; un-awaited promises keep mutation sites) while keeping the depth (`MAX_VALUE_DEPTH`), circularity,
the await-hinting rejection in BOTH modes (deliberate — JS-parity pass-through was plain-objects-only, blocked-property, and data-only checks; un-awaited promises keep
considered and skipped to preserve the nudge). The HOST boundary (final result, the await-hinting rejection in BOTH modes (deliberate — JS-parity pass-through was
tool-call arguments, `JSON.stringify`, tool-result intake) uses the default mode and considered and skipped to preserve the nudge). The HOST boundary (final result,
still serializes JSON forms (Date → ISO, RegExp/Map/Set → `{}`); host instances met on tool-call arguments, `JSON.stringify`, tool-result intake) uses the default mode and
the preserving path are defensively wrapped into sandbox equivalents. Ripple: the still serializes JSON forms (Date → ISO, RegExp/Map/Set → `{}`); host instances met on
`Object.*` helpers treat sandbox values as empty objects (`Object.keys(map)``[]`, the preserving path are defensively wrapped into sandbox equivalents. Ripple: the
assign sources contribute nothing, hasOwn → false — JS has no own enumerable props `Object.*` helpers treat sandbox values as empty objects (`Object.keys(map)``[]`,
there), so interpreter internals (`.map`/`.time`/`.regex`) can never leak; the assign sources contribute nothing, hasOwn → false — JS has no own enumerable props
template-literal sandbox carve-out collapsed into `boundedData`. Object/array spread there), so interpreter internals (`.map`/`.time`/`.regex`) can never leak; the
already preserved instances (reference copies, no checkpoint) — now tested. template-literal sandbox carve-out collapsed into `boundedData`. Object/array spread
- **Console formatting**: `formatConsoleArgument` is total and deep already preserved instances (reference copies, no checkpoint) — now tested.
(`formatConsoleValue`): numbers render via `String` (`NaN`/`Infinity`/`-Infinity` - **Console formatting**: `formatConsoleArgument` is total and deep
literally — never the JSON `null`; finite numbers match their JSON form), nested (`formatConsoleValue`): numbers render via `String` (`NaN`/`Infinity`/`-Infinity`
strings are JSON-quoted, sandbox values keep their friendly forms at ANY depth (ISO literally — never the JSON `null`; finite numbers match their JSON form), nested
date, `/regex/flags`, `Map(n) [...]`, `Set(n) [...]`), opaque references become strings are JSON-quoted, sandbox values keep their friendly forms at ANY depth (ISO
in-place `[CodeMode reference]` markers instead of collapsing the whole argument, date, `/regex/flags`, `Map(n) [...]`, `Set(n) [...]`), opaque references become
cycles render `[Circular]` (reachable via Map/Set members, which mutation never in-place `[CodeMode reference]` markers instead of collapsing the whole argument,
checkpoints), and depth beyond `MAX_CONSOLE_DEPTH = 32` (fixed constant, not a knob) cycles render `[Circular]` (reachable via Map/Set members, which mutation never
degrades to `…` — console can no longer fail a program. `console.table` guards with checkpoints), and depth beyond `MAX_CONSOLE_DEPTH = 32` (fixed constant, not a knob)
`containsOpaqueReference` (sandbox cells render, e.g. ISO dates) and its row/cell degrades to `…` — console can no longer fail a program. `console.table` guards with
walkers treat sandbox values as scalar cells. `containsOpaqueReference` (sandbox cells render, e.g. ISO dates) and its row/cell
- **Prose**: the instructions Syntax not-supported line dropped its `instanceof walkers treat sandbox values as scalar cells.
Error`/splice mentions (nothing else reworded); README updated (checkpoint - **Prose**: the instructions Syntax not-supported line dropped its `instanceof
preservation vs boundary serialization, error values/`instanceof`, new array/string Error`/splice mentions (nothing else reworded); README updated (checkpoint
methods, regex-failure behavior); `supportedSyntaxMessage` left untouched (it lists preservation vs boundary serialization, error values/`instanceof`, new array/string
supported syntax, was already non-exhaustive, and stays accurate). methods, regex-failure behavior); `supportedSyntaxMessage` left untouched (it lists
- **Tests**: package suite 169 → 209 (parity: Error/instanceof + real-JS error-name supported syntax, was already non-exhaustive, and stays accurate).
coverage, splice/fill/copyWithin/keys/values/entries, localeCompare/normalize/trim-alias - **Tests**: package suite 169 → 209 (parity: Error/instanceof + real-JS error-name
describes; stdlib: checkpoint survival incl. tool-arg boundary pinning, stdlib coverage, splice/fill/copyWithin/keys/values/entries, localeCompare/normalize/trim-alias
`instanceof`, regex-message assertions; codemode: NaN/Infinity + nested/cyclic console describes; stdlib: checkpoint survival incl. tool-arg boundary pinning, stdlib
rendering, table cells, caught-tool-failure `instanceof`); adapter suites unchanged `instanceof`, regex-message assertions; codemode: NaN/Infinity + nested/cyclic console
(34 + 16, green); both packages `tsgo --noEmit` clean. rendering, table cells, caught-tool-failure `instanceof`); adapter suites unchanged
(34 + 16, green); both packages `tsgo --noEmit` clean.
**Truncation layering — CodeMode truncation off in OpenCode** (user direction; resolves the **Truncation layering — CodeMode truncation off in OpenCode** (user direction; resolves the
§4 outer-truncation item the OPPOSITE way from "kill the outer one"): §4 outer-truncation item the OPPOSITE way from "kill the outer one"):
- `maxOutputBytes` lost its 32,000 default and now behaves exactly like the other two
limits: absent = no truncation. All three limits are uniformly no-default — budgets are - `maxOutputBytes` lost its 32,000 default and now behaves exactly like the other two
host policy. `ResolvedExecutionLimits.maxOutputBytes` is `number | undefined`; limits: absent = no truncation. All three limits are uniformly no-default — budgets are
`boundOutput` only runs when the host set the limit. Explicit values validate as before host policy. `ResolvedExecutionLimits.maxOutputBytes` is `number | undefined`;
(safe integer ≥ 0). `boundOutput` only runs when the host set the limit. Explicit values validate as before
- OpenCode continues to pass NO limits, which now also means no CodeMode truncation. (safe integer ≥ 0).
`execute` is a normal `Tool.define` tool, so OpenCode's native tool-output truncation - OpenCode continues to pass NO limits, which now also means no CodeMode truncation.
applies with no special-casing — verified by tracing `wrap()` (`tool.ts:130-144`, `execute` is a normal `Tool.define` tool, so OpenCode's native tool-output truncation
50KB/2000-line thresholds in `truncate.ts`, full output dumped to a file under applies with no special-casing — verified by tracing `wrap()` (`tool.ts:130-144`,
`tool-output/`): the `metadata.truncated` self-truncation exemption never fires for 50KB/2000-line thresholds in `truncate.ts`, full output dumped to a file under
`execute` (its metadata never sets that key). One truncation layer, the host's — and it `tool-output/`): the `metadata.truncated` self-truncation exemption never fires for
is the richer one (file dump + explore/grep hint vs an inline marker). `execute` (its metadata never sets that key). One truncation layer, the host's — and it
- Hosts without their own output bounding set `maxOutputBytes` explicitly; README table is the richer one (file dump + explore/grep hint vs an inline marker).
and prose updated, adapter comment rewritten. Tests: codemode +1 (absent limit → 100KB - Hosts without their own output bounding set `maxOutputBytes` explicitly; README table
value + 50KB log line pass through unbounded, `truncated` undefined); the adapter test and prose updated, adapter comment rewritten. Tests: codemode +1 (absent limit → 100KB
that relied on the old default now asserts the oversized result reaches the shared value + 50KB log line pass through unbounded, `truncated` undefined); the adapter test
wrapper un-truncated. Suites: 210 + 50, tsgo clean both. that relied on the old default now asserts the oversized result reaches the shared
wrapper un-truncated. Suites: 210 + 50, tsgo clean both.
**Docs polish** (post-API-review): stale `DiscoveryOptions` JSDoc fixed (claimed default **Docs polish** (post-API-review): stale `DiscoveryOptions` JSDoc fixed (claimed default
4,000 and alphabetical cheapest-first — now 2,000 and round-robin, matching Fix 8/9 reality) 4,000 and alphabetical cheapest-first — now 2,000 and round-robin, matching Fix 8/9 reality)
@ -837,132 +856,137 @@ regular dependency; hosts depend on it themselves because the API surface is Eff
**Registry promotion + permission-aware catalog** (the "promote to a proper tool service" **Registry promotion + permission-aware catalog** (the "promote to a proper tool service"
restructure; fixes the §4 permission-advertising bug): restructure; fixes the §4 permission-advertising bug):
- **The adapter moved** `src/session/code-mode.ts``src/tool/code-mode.ts` and is now a
registry-resident tool service on the TaskTool precedent: `CodeModeTool = - **The adapter moved** `src/session/code-mode.ts``src/tool/code-mode.ts` and is now a
Tool.define(CODE_MODE_TOOL, ...)` whose init depends on `MCP.Service`, `Agent.Service`, registry-resident tool service on the TaskTool precedent: `CodeModeTool =
and `Session.Service`. It is yielded in `ToolRegistry.layer`, gated into `builtin` by Tool.define(CODE_MODE_TOOL, ...)` whose init depends on `MCP.Service`, `Agent.Service`,
`flags.experimentalCodeMode` (like the lsp/plan experiments), and `MCP.node` joined the and `Session.Service`. It is yielded in `ToolRegistry.layer`, gated into `builtin` by
registry's `node.deps` (`MCP.node` has no ToolRegistry dependency, so no cycle). The `flags.experimentalCodeMode` (like the lsp/plan experiments), and `MCP.node` joined the
session-level special-casing in `session/tools.ts` (ad-hoc `SessionCodeMode.define` + registry's `node.deps` (`MCP.node` has no ToolRegistry dependency, so no cycle). The
append) is deleted; the early return that suppresses raw per-MCP registration when the session-level special-casing in `session/tools.ts` (ad-hoc `SessionCodeMode.define` +
flag is on stays session-side, keyed on the same flag+tool-count condition. append) is deleted; the early return that suppresses raw per-MCP registration when the
- **Enablement** lives in `ToolRegistry.tools()` next to the WebSearchTool check: the MCP flag is on stays session-side, keyed on the same flag+tool-count condition.
tool count is consulted once (an Effect) before the synchronous filter, and code mode - **Enablement** lives in `ToolRegistry.tools()` next to the WebSearchTool check: the MCP
passes the predicate iff `flags.experimentalCodeMode` && count > 0. tool count is consulted once (an Effect) before the synchronous filter, and code mode
- **Description split on the `describeTask` precedent**: the tool's static base passes the predicate iff `flags.experimentalCodeMode` && count > 0.
description is a two-line summary; `describeCodeMode(agent)` in `registry.tools()` - **Description split on the `describeTask` precedent**: the tool's static base
appends the full CodeMode instructions (workflow/rules/syntax + grouped catalog, description is a two-line summary; `describeCodeMode(agent)` in `registry.tools()`
`catalogInstructions` in the adapter) at the same composition point as task — so appends the full CodeMode instructions (workflow/rules/syntax + grouped catalog,
`plugin.trigger("tool.definition")` sees the base description first. `catalogInstructions` in the adapter) at the same composition point as task — so
- **Permission-aware catalog + dispatch** (the bug fix): the visibility predicate from `plugin.trigger("tool.definition")` sees the base description first.
`llm/request.ts` `resolveTools` is hoisted to `Permission.visibleTools(tools, ruleset)` - **Permission-aware catalog + dispatch** (the bug fix): the visibility predicate from
(a record filter over `Permission.disabled` — only a hard `deny` with pattern `"*"` `llm/request.ts` `resolveTools` is hoisted to `Permission.visibleTools(tools, ruleset)`
hides a tool; ask-level rules stay fully visible and prompt at call time) and (a record filter over `Permission.disabled` — only a hard `deny` with pattern `"*"`
`resolveTools` now uses it, so the two paths cannot drift. `describeCodeMode` filters hides a tool; ask-level rules stay fully visible and prompt at call time) and
with the merged agent+session ruleset that `SessionTools.resolve` passes into the `resolveTools` now uses it, so the two paths cannot drift. `describeCodeMode` filters
registry before building the catalog/search index; `execute` rebuilds the runtime per with the merged agent+session ruleset that `SessionTools.resolve` passes into the
execution from a fresh, filtered `mcp.tools()` snapshot using the same merged ruleset registry before building the catalog/search index; `execute` rebuilds the runtime per
(`Agent.get(ctx.agent)` + `Session.get(ctx.sessionID)`, matching the merge execution from a fresh, filtered `mcp.tools()` snapshot using the same merged ruleset
`SessionTools.context` wires into `ctx.ask`) — a denied tool is not dispatchable (`Agent.get(ctx.agent)` + `Session.get(ctx.sessionID)`, matching the merge
even if the model guesses its name and yields the normal unknown-tool diagnostic. `SessionTools.context` wires into `ctx.ask`) — a denied tool is not dispatchable
Documented gap (out of scope by design): per-message `user.tools[key] === false` arrives even if the model guesses its name and yields the normal unknown-tool diagnostic.
at request-prep after descriptions are built and has no child-call equivalent. Documented gap (out of scope by design): per-message `user.tools[key] === false` arrives
- **Preserved behavior**: cancellation race + pre-aborted-signal guard, `toSandboxResult` at request-prep after descriptions are built and has no child-call equivalent.
unwrap order, attachment accumulation, `CODE_MODE_TOOL` at all title sites, no execution - **Preserved behavior**: cancellation race + pre-aborted-signal guard, `toSandboxResult`
limits (native truncation only), `displayInput`, per-child `ctx.ask` gating (now wired unwrap order, attachment accumulation, `CODE_MODE_TOOL` at all title sites, no execution
through `Tool.Context` exactly like every registry tool). limits (native truncation only), `displayInput`, per-child `ctx.ask` gating (now wired
- **Explicit non-goal**: memoizing the catalog builder keyed on (ToolsChanged generation, through `Tool.Context` exactly like every registry tool).
permission ruleset) was considered and deliberately skipped — the per-turn rebuild is - **Explicit non-goal**: memoizing the catalog builder keyed on (ToolsChanged generation,
cheap (grouping + string rendering); revisit only if profiling shows it matters. permission ruleset) was considered and deliberately skipped — the per-turn rebuild is
- **Tests**: the two adapter suites moved to `test/tool/{code-mode,code-mode-integration} cheap (grouping + string rendering); revisit only if profiling shows it matters.
.test.ts` (mocked `MCP.Service`/`Agent.Service`/`Session.Service` replacing the direct - **Tests**: the two adapter suites moved to `test/tool/{code-mode,code-mode-integration}
`define(...)` construction; description assertions target `catalogInstructions`, the .test.ts` (mocked `MCP.Service`/`Agent.Service`/`Session.Service` replacing the direct
registry's composition input) and gained permission coverage: deny excluded from `define(...)` construction; description assertions target `catalogInstructions`, the
catalog/search, ask-level stays visible and callable, denied tool undispatchable registry's composition input) and gained permission coverage: deny excluded from
(unknown-tool diagnostic), `Permission.visibleTools` semantics. `test/tool/ catalog/search, ask-level stays visible and callable, denied tool undispatchable
registry.test.ts` gained four registry-level tests: registered with flag+MCP tools, (unknown-tool diagnostic), `Permission.visibleTools` semantics. `test/tool/
excluded without MCP tools, excluded with flag off, and deny/ask catalog filtering registry.test.ts` gained four registry-level tests: registered with flag+MCP tools,
through `registry.tools()`. Suites: 43 + 16 adapter tests, 16 registry tests, all green. excluded without MCP tools, excluded with flag off, and deny/ask catalog filtering
through `registry.tools()`. Suites: 43 + 16 adapter tests, 16 registry tests, all green.
**Shared MCP invocation middle (`McpInvoke.invoke`)** (closes the §4 "plugin hooks skip **Shared MCP invocation middle (`McpInvoke.invoke`)** (closes the §4 "plugin hooks skip
child calls" gap): child calls" gap):
- `packages/opencode/src/mcp/invoke.ts` extracts the duplicated "invoke an MCP tool"
middle into one shared `McpInvoke.invoke(input)`: plugin `tool.execute.before` hook → - `packages/opencode/src/mcp/invoke.ts` extracts the duplicated "invoke an MCP tool"
permission ask (`{ permission: key, patterns: ["*"], always: ["*"] }` via the caller's middle into one shared `McpInvoke.invoke(input)`: plugin `tool.execute.before` hook →
`ctx.ask`) → dispatch through the ai-sdk tool's execute inside the `Tool.execute` permission ask (`{ permission: key, patterns: ["*"], always: ["*"] }` via the caller's
tracing span (`tool.name`/`tool.call_id`/`session.id`/`message.id` attributes) → `ctx.ask`) → dispatch through the ai-sdk tool's execute inside the `Tool.execute`
plugin `tool.execute.after` hook. It returns the RAW result the ai-sdk execute tracing span (`tool.name`/`tool.call_id`/`session.id`/`message.id` attributes) →
resolved with; each caller keeps its own shaping edge — the legacy per-MCP loop in plugin `tool.execute.after` hook. It returns the RAW result the ai-sdk execute
`SessionTools.resolve` applies its existing model-facing shaping/truncation, code resolved with; each caller keeps its own shaping edge — the legacy per-MCP loop in
mode applies `toSandboxResult`. It lives under `src/mcp/` because both callers `SessionTools.resolve` applies its existing model-facing shaping/truncation, code
already depend on MCP and the function is about invoking an MCP-backed ai-sdk tool, mode applies `toSandboxResult`. It lives under `src/mcp/` because both callers
not about sessions or code mode. already depend on MCP and the function is about invoking an MCP-backed ai-sdk tool,
- **After-hook payload**: fired inside `McpInvoke.invoke` with the raw MCP result — not about sessions or code mode.
which is exactly what the legacy loop always passed (the raw `CallToolResult`, not - **After-hook payload**: fired inside `McpInvoke.invoke` with the raw MCP result —
the shaped `{title, output, metadata}`), so legacy behavior is preserved bit-for-bit which is exactly what the legacy loop always passed (the raw `CallToolResult`, not
and the hook payload cannot drift between callers. No callback/edge-firing design the shaped `{title, output, metadata}`), so legacy behavior is preserved bit-for-bit
was needed. and the hook payload cannot drift between callers. No callback/edge-firing design
- **Synthetic child callID**: code-mode child calls pass `${parentCallID}/${n}` as the was needed.
hook/span callID (`parentCallID` = the `execute` call's `ctx.callID`, falling back to - **Synthetic child callID**: code-mode child calls pass `${parentCallID}/${n}` as the
the entry key; `n` = per-execution counter starting at 1, shared across all child hook/span callID (`parentCallID` = the `execute` call's `ctx.callID`, falling back to
calls in one program). callID is an opaque string — nothing parses it. The ai-sdk the entry key; `n` = per-execution counter starting at 1, shared across all child
`toolCallId` (`options.toolCallId`) stays each caller's existing value calls in one program). callID is an opaque string — nothing parses it. The ai-sdk
(`ctx.callID ?? entry.key` for code mode). `toolCallId` (`options.toolCallId`) stays each caller's existing value
- **Child-scoped hook failures**: `CodeModeTool` (which now also yields (`ctx.callID ?? entry.key` for code mode).
`Plugin.Service`) wraps the whole child call — hooks, ask, dispatch — in - **Child-scoped hook failures**: `CodeModeTool` (which now also yields
`toCatchable` (the generalization of the old `askPermission` catchCause), so a plugin `Plugin.Service`) wraps the whole child call — hooks, ask, dispatch — in
hook failure fails ONLY that child call as a catchable in-program `toolError`; other `toCatchable` (the generalization of the old `askPermission` catchCause), so a plugin
calls in the same program keep running and interruption still propagates as hook failure fails ONLY that child call as a catchable in-program `toolError`; other
interruption. Legacy semantics unchanged: a hook failure fails the tool call. calls in the same program keep running and interruption still propagates as
- **Tests**: `test/tool/code-mode.test.ts` +2 (child calls fire before/after with the interruption. Legacy semantics unchanged: a hook failure fails the tool call.
MCP key and `parent/1`, `parent/2` ids, after hook carries the raw MCP result; a - **Tests**: `test/tool/code-mode.test.ts` +2 (child calls fire before/after with the
failing before hook is caught in-program, gates dispatch, and leaves the outer MCP key and `parent/1`, `parent/2` ids, after hook carries the raw MCP result; a
execute ok) — both code-mode harnesses gained a `Plugin.Service` mock (pass-through failing before hook is caught in-program, gates dispatch, and leaves the outer
trigger by default, overridable). New `test/session/tools.test.ts` (3 tests) pins execute ok) — both code-mode harnesses gained a `Plugin.Service` mock (pass-through
`SessionTools.resolve` at the real-registry seam (LayerNode.compile, fake MCP layer): trigger by default, overridable). New `test/session/tools.test.ts` (3 tests) pins
flag on + MCP tools → `execute` present, raw MCP keys suppressed; flag off → raw `SessionTools.resolve` at the real-registry seam (LayerNode.compile, fake MCP layer):
keys present, `execute` absent; and the legacy raw-MCP execute fires before/after flag on + MCP tools → `execute` present, raw MCP keys suppressed; flag off → raw
hooks keyed by the ai-sdk toolCallId with the raw result payload. Suites: adapter keys present, `execute` absent; and the legacy raw-MCP execute fires before/after
45 + 16, session/tool/permission all green; this package untouched (211 pass). hooks keyed by the ai-sdk toolCallId with the raw result payload. Suites: adapter
45 + 16, session/tool/permission all green; this package untouched (211 pass).
**Signature rendering + compound-assignment parity fixes** (externally reported, both **Signature rendering + compound-assignment parity fixes** (externally reported, both
verified real with failing tests before fixing): verified real with failing tests before fixing):
- **Non-identifier property names in rendered signatures** (`src/tool.ts`): `renderSchema`
emitted raw property names, so schema properties like `foo-bar`/`@type`/`x.y`/`123` - **Non-identifier property names in rendered signatures** (`src/tool.ts`): `renderSchema`
rendered invalid TypeScript (`{ foo-bar?: string }`). Fixed with a `renderKey` helper — emitted raw property names, so schema properties like `foo-bar`/`@type`/`x.y`/`123`
bare identifiers stay bare, everything else is `JSON.stringify`-quoted — applied in the rendered invalid TypeScript (`{ foo-bar?: string }`). Fixed with a `renderKey` helper —
single `field` closure both the compact and pretty renderings share. The bare identifiers stay bare, everything else is `JSON.stringify`-quoted — applied in the
`identifierSegment` regex now lives in `tool.ts` (exported) and `tool-runtime.ts`'s single `field` closure both the compact and pretty renderings share. The
bracket-notation `toolExpression` imports it: one source of truth for "is this a bare `identifierSegment` regex now lives in `tool.ts` (exported) and `tool-runtime.ts`'s
identifier" across object keys and tool paths. Tests: `signature.test.ts` +4 (compact, bracket-notation `toolExpression` imports it: one source of truth for "is this a bare
pretty with JSDoc on a quoted key, JSON Schema input+output, Effect Schema struct). identifier" across object keys and tool paths. Tests: `signature.test.ts` +4 (compact,
- **Numeric schema unions keep their real alternatives** (`src/tool.ts`): the old pretty with JSDoc on a quoted key, JSON Schema input+output, Effect Schema struct).
`anyOf`/`oneOf` renderer collapsed any union containing `{ type: "number" }` to just - **Numeric schema unions keep their real alternatives** (`src/tool.ts`): the old
`number`, dropping real JSON Schema alternatives (`string | number`, `number | null`, `anyOf`/`oneOf` renderer collapsed any union containing `{ type: "number" }` to just
etc.). The collapse is now restricted to Effect's number-schema artifact `number`, dropping real JSON Schema alternatives (`string | number`, `number | null`,
(`number | "NaN" | "Infinity" | "-Infinity"`, emitted as single-value string enums), etc.). The collapse is now restricted to Effect's number-schema artifact
while raw JSON Schema unions render every branch. Tests: `signature.test.ts` +3. (`number | "NaN" | "Infinity" | "-Infinity"`, emitted as single-value string enums),
- **Compound assignment now matches binary-operator semantics** (`src/codemode.ts`): while raw JSON Schema unions render every branch. Tests: `signature.test.ts` +3.
`applyCompoundAssignment` did raw JS ops on interpreter wrapper objects, so `x += y` - **Compound assignment now matches binary-operator semantics** (`src/codemode.ts`):
diverged from `x = x + y` (sandbox Date `d += 1` produced `"[object Object]1"`; `applyCompoundAssignment` did raw JS ops on interpreter wrapper objects, so `x += y`
`d -= 400` gave `NaN` instead of epoch arithmetic). The operator table + coercion moved diverged from `x = x + y` (sandbox Date `d += 1` produced `"[object Object]1"`;
verbatim out of `evaluateBinaryExpression` into a shared `applyBinaryOperator`; `d -= 400` gave `NaN` instead of epoch arithmetic). The operator table + coercion moved
compound assignment validates against a `compoundOperators` set (`+=``>>>=`) and verbatim out of `evaluateBinaryExpression` into a shared `applyBinaryOperator`;
dispatches through it (`operator.slice(0, -1)`). Logical assignments (`&&=`/`||=`/`??=`) compound assignment validates against a `compoundOperators` set (`+=``>>>=`) and
keep their separate short-circuit path (`evaluateLogicalAssignment`), and both dispatches through it (`operator.slice(0, -1)`). Logical assignments (`&&=`/`||=`/`??=`)
assignment call sites still wrap results in `boundedData`. Deliberate side effect: keep their separate short-circuit path (`evaluateLogicalAssignment`), and both
compound assignment now rejects opaque references, consistent with binary operators. assignment call sites still wrap results in `boundedData`. Deliberate side effect:
Tests: `parity.test.ts` +5 (Date `+=` concat parity, Date `-=`/`/=` epoch parity, compound assignment now rejects opaque references, consistent with binary operators.
string `+=` object/array, member-target compound, 13-case operator sweep vs real JS). Tests: `parity.test.ts` +5 (Date `+=` concat parity, Date `-=`/`/=` epoch parity,
Package suite: 220 pass. string `+=` object/array, member-target compound, 13-case operator sweep vs real JS).
Package suite: 220 pass.
--- ---
## 4. Remaining work (detailed TODO) ## 4. Remaining work (detailed TODO)
### Next DSL-expansion pass (done — see the DSL-expansion pass entry in §3) ### Next DSL-expansion pass (done — see the DSL-expansion pass entry in §3)
Batch these together — per user direction: important, but deliberately deferred to one Batch these together — per user direction: important, but deliberately deferred to one
focused interpreter-surface pass rather than picked off piecemeal. focused interpreter-surface pass rather than picked off piecemeal.
- [x] Medium-tier JS parity items deferred from the original audit: caught errors are plain - [x] Medium-tier JS parity items deferred from the original audit: caught errors are plain
`{ name, message }` objects, not `instanceof Error` (and `Error` isn't a value — `{ name, message }` objects, not `instanceof Error` (and `Error` isn't a value —
`x instanceof Error` is unsupported syntax); `splice` (still a `x instanceof Error` is unsupported syntax); `splice` (still a
@ -981,6 +1005,7 @@ focused interpreter-surface pass rather than picked off piecemeal.
(`console.log({ m: map })`) — could deep-format instead. (`console.log({ m: map })`) — could deep-format instead.
### Next iteration: text-result handling (deliberate follow-up, user-directed) ### Next iteration: text-result handling (deliberate follow-up, user-directed)
- [ ] Revisit how MCP text results reach the program. Today: `structuredContent` when the - [ ] Revisit how MCP text results reach the program. Today: `structuredContent` when the
server sends it, else joined text as a plain string (the program JSON.parses it, server sends it, else joined text as a plain string (the program JSON.parses it,
guided by a workflow step). Considered and deferred: (a) conservative boundary guided by a workflow step). Considered and deferred: (a) conservative boundary
@ -992,6 +1017,7 @@ focused interpreter-surface pass rather than picked off piecemeal.
revisit once real usage shows which failure modes matter. revisit once real usage shows which failure modes matter.
### Next iteration: stdlib surface (prioritized) ### Next iteration: stdlib surface (prioritized)
Current instructions say "usual Array/String/Object/Math/JSON methods," but the interpreter is Current instructions say "usual Array/String/Object/Math/JSON methods," but the interpreter is
intentionally a subset. Keep CodeMode focused on orchestration and data shaping, not a full host intentionally a subset. Keep CodeMode focused on orchestration and data shaping, not a full host
runtime, but close the high-friction gaps models are likely to reach for. runtime, but close the high-friction gaps models are likely to reach for.
@ -1028,7 +1054,9 @@ Explicit non-goals for now: `structuredClone`, `WeakMap`/`WeakSet`, and timers
orchestration use case. orchestration use case.
### Wiring-review findings (subagent code review of the OpenCode integration, triaged) ### Wiring-review findings (subagent code review of the OpenCode integration, triaged)
Pre-PR fixes (user-approved cut): Pre-PR fixes (user-approved cut):
- [x] **Cancellation does not interrupt the interpreter** — the no-limits rationale claimed - [x] **Cancellation does not interrupt the interpreter** — the no-limits rationale claimed
"user cancel interrupts the execution fiber," but `tools.ts` runs tools via "user cancel interrupts the execution fiber," but `tools.ts` runs tools via
`run.promise``Effect.runPromise` (`effect/bridge.ts:64-66`) with NO abort wiring; `run.promise``Effect.runPromise` (`effect/bridge.ts:64-66`) with NO abort wiring;
@ -1073,6 +1101,7 @@ Pre-PR fixes (user-approved cut):
`title: "execute"` sites in `code-mode.ts` now reference `CODE_MODE_TOOL`. `title: "execute"` sites in `code-mode.ts` now reference `CODE_MODE_TOOL`.
Post-MVP (logged, not blocking an experimental flag): Post-MVP (logged, not blocking an experimental flag):
- [x] **Plugin `tool.execute.before/after` hooks skip child calls** — legacy MCP - [x] **Plugin `tool.execute.before/after` hooks skip child calls** — legacy MCP
registration fires them per tool (`tools.ts:419-441`); under code mode only the registration fires them per tool (`tools.ts:419-441`); under code mode only the
outer `execute` fires them, so auditing/intercepting plugins silently lose MCP outer `execute` fires them, so auditing/intercepting plugins silently lose MCP
@ -1105,6 +1134,7 @@ Post-MVP (logged, not blocking an experimental flag):
names that are no longer directly callable under code mode. names that are no longer directly callable under code mode.
### Backlog / loose ends (non-blocking, any order) ### Backlog / loose ends (non-blocking, any order)
- [ ] `evaluateUpdateExpression` (`++`/`--`) still uses raw `Number(current)`, so `d++` on a - [ ] `evaluateUpdateExpression` (`++`/`--`) still uses raw `Number(current)`, so `d++` on a
sandbox Date yields `NaN` where `d += 1` now uses epoch semantics (and real JS `d++` sandbox Date yields `NaN` where `d += 1` now uses epoch semantics (and real JS `d++`
would give epoch+0 numeric). Pre-existing, out of scope of the compound-assignment would give epoch+0 numeric). Pre-existing, out of scope of the compound-assignment
@ -1141,7 +1171,7 @@ Post-MVP (logged, not blocking an experimental flag):
## 5. Context and gotchas for whoever picks this up ## 5. Context and gotchas for whoever picks this up
- **Motivating failure (why forgiving semantics + prompting matter):** in a real transcript, - **Motivating failure (why forgiving semantics + prompting matter):** in a real transcript,
the model wrote `me.result?.login ?? me.result` where the tool result was a JSON *string* the model wrote `me.result?.login ?? me.result` where the tool result was a JSON _string_
the old strict interpreter threw (`String property 'login' is not available`); then the the old strict interpreter threw (`String property 'login' is not available`); then the
model returned a raw 105KB payload, which native truncation dumped to a file, costing a model returned a raw 105KB payload, which native truncation dumped to a file, costing a
subagent round-trip to extract one number. Interpreter forgiveness stops the crashes; subagent round-trip to extract one number. Interpreter forgiveness stops the crashes;

File diff suppressed because it is too large Load diff

View file

@ -1,10 +1,4 @@
export { export { ToolError, CodeMode, ExecuteInputSchema, ExecuteResultSchema, toolError } from "./codemode.js"
ToolError,
CodeMode,
ExecuteInputSchema,
ExecuteResultSchema,
toolError,
} from "./codemode.js"
export { Tool } from "./tool.js" export { Tool } from "./tool.js"
export type { Definition as ToolDefinition, JsonSchema, ToolSchema } from "./tool.js" export type { Definition as ToolDefinition, JsonSchema, ToolSchema } from "./tool.js"
export type { ToolCallEnded, ToolCallHooks } from "./tool-runtime.js" export type { ToolCallEnded, ToolCallHooks } from "./tool-runtime.js"

View file

@ -21,11 +21,16 @@ export type HostTools<R = never> = {
export type Services<Tools> = Tools extends (...args: Array<unknown>) => Effect.Effect<unknown, unknown, infer R> export type Services<Tools> = Tools extends (...args: Array<unknown>) => Effect.Effect<unknown, unknown, infer R>
? R ? R
: Tools extends { readonly _tag: "CodeModeTool"; readonly run: (input: unknown) => Effect.Effect<unknown, unknown, infer R> } : Tools extends {
readonly _tag: "CodeModeTool"
readonly run: (input: unknown) => Effect.Effect<unknown, unknown, infer R>
}
? R ? R
: Tools extends object : Tools extends object
? string extends keyof Tools ? never : Services<Tools[keyof Tools]> ? string extends keyof Tools
: never ? never
: Services<Tools[keyof Tools]>
: never
/** Minimal audit record retained for each admitted tool call. */ /** Minimal audit record retained for each admitted tool call. */
export type ToolCall = { export type ToolCall = {
@ -68,11 +73,13 @@ export type SafeObject = Record<string, unknown>
const reservedNamespace = "$codemode" const reservedNamespace = "$codemode"
const defaultMaxInlineCatalogTokens = 2_000 const defaultMaxInlineCatalogTokens = 2_000
const defaultSearchLimit = 10 const defaultSearchLimit = 10
const searchSignature = "tools.$codemode.search({ query?: string, namespace?: string, limit?: number }): Promise<{ items: Array<{ path: string; description: string; signature: string }>; total: number }>" const searchSignature =
"tools.$codemode.search({ query?: string, namespace?: string, limit?: number }): Promise<{ items: Array<{ path: string; description: string; signature: string }>; total: number }>"
const toolExpression = (path: string) => const toolExpression = (path: string) =>
"tools" + path "tools" +
path
.split(".") .split(".")
.map((segment) => identifierSegment.test(segment) ? `.${segment}` : `[${JSON.stringify(segment)}]`) .map((segment) => (identifierSegment.test(segment) ? `.${segment}` : `[${JSON.stringify(segment)}]`))
.join("") .join("")
export class ToolReference { export class ToolReference {
@ -88,7 +95,12 @@ const MAX_VALUE_DEPTH = 32
export class ToolRuntimeError extends Error { export class ToolRuntimeError extends Error {
constructor( constructor(
readonly kind: "UnknownTool" | "InvalidToolInput" | "InvalidToolOutput" | "InvalidDataValue" | "ToolCallLimitExceeded", readonly kind:
| "UnknownTool"
| "InvalidToolInput"
| "InvalidToolOutput"
| "InvalidDataValue"
| "ToolCallLimitExceeded",
message: string, message: string,
readonly suggestions: ReadonlyArray<string> = [], readonly suggestions: ReadonlyArray<string> = [],
) { ) {
@ -131,7 +143,13 @@ export const isBlockedMember = (name: string): boolean => blockedMemberNames.has
export const copyIn = (value: unknown, label: string, preserveSandboxValues = false): unknown => export const copyIn = (value: unknown, label: string, preserveSandboxValues = false): unknown =>
copyBounded(value, label, 0, new Set(), preserveSandboxValues) copyBounded(value, label, 0, new Set(), preserveSandboxValues)
const copyBounded = (value: unknown, label: string, depth: number, seen: Set<object>, preserveSandboxValues: boolean): unknown => { const copyBounded = (
value: unknown,
label: string,
depth: number,
seen: Set<object>,
preserveSandboxValues: boolean,
): unknown => {
if (depth > MAX_VALUE_DEPTH) { if (depth > MAX_VALUE_DEPTH) {
throw new ToolRuntimeError("InvalidDataValue", `${label} exceeds the maximum value depth of ${MAX_VALUE_DEPTH}.`) throw new ToolRuntimeError("InvalidDataValue", `${label} exceeds the maximum value depth of ${MAX_VALUE_DEPTH}.`)
} }
@ -166,7 +184,12 @@ const copyBounded = (value: unknown, label: string, depth: number, seen: Set<obj
// Intra-sandbox checkpoints keep sandbox value instances alive as leaves; their contents // Intra-sandbox checkpoints keep sandbox value instances alive as leaves; their contents
// are never walked here (Map/Set members are validated where mutation happens, and the // are never walked here (Map/Set members are validated where mutation happens, and the
// real boundary still serializes them below). // real boundary still serializes them below).
if (value instanceof SandboxDate || value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet) { if (
value instanceof SandboxDate ||
value instanceof SandboxRegExp ||
value instanceof SandboxMap ||
value instanceof SandboxSet
) {
return value return value
} }
// Host instances cannot normally reach an intra-sandbox checkpoint (tool results cross // Host instances cannot normally reach an intra-sandbox checkpoint (tool results cross
@ -197,8 +220,12 @@ const copyBounded = (value: unknown, label: string, depth: number, seen: Set<obj
return Number.isFinite(value.getTime()) ? value.toISOString() : null return Number.isFinite(value.getTime()) ? value.toISOString() : null
} }
if ( if (
value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet || value instanceof SandboxRegExp ||
value instanceof RegExp || value instanceof Map || value instanceof Set value instanceof SandboxMap ||
value instanceof SandboxSet ||
value instanceof RegExp ||
value instanceof Map ||
value instanceof Set
) { ) {
return Object.create(null) as SafeObject return Object.create(null) as SafeObject
} }
@ -250,7 +277,10 @@ export const copyOut = (value: unknown, undefinedAsNull = false): unknown => {
return value return value
} }
const definitions = <R>(tools: HostTools<R>, path: ReadonlyArray<string> = []): Array<{ path: string; definition: Definition<R> }> => { const definitions = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string> = [],
): Array<{ path: string; definition: Definition<R> }> => {
const entries: Array<{ path: string; definition: Definition<R> }> = [] const entries: Array<{ path: string; definition: Definition<R> }> = []
for (const [name, value] of Object.entries(tools)) { for (const [name, value] of Object.entries(tools)) {
const next = [...path, name] const next = [...path, name]
@ -342,8 +372,11 @@ const toSearchEntry = <R>(path: string, definition: Definition<R>, description:
path, path,
definition.description, definition.description,
...inputProperties(definition).flatMap(({ name, description: property }) => ...inputProperties(definition).flatMap(({ name, description: property }) =>
property === undefined ? [name] : [name, property]), property === undefined ? [name] : [name, property],
].join("\n").toLowerCase(), ),
]
.join("\n")
.toLowerCase(),
}) })
/** The runtime search index over every described tool. Search is always registered. */ /** The runtime search index over every described tool. Search is always registered. */
@ -396,7 +429,8 @@ export const discoveryPlan = <R>(
namespace, namespace,
picked: new Set<ToolDescription>(), picked: new Set<ToolDescription>(),
queue: [...group].sort( queue: [...group].sort(
(left, right) => estimate(catalogLine(left)) - estimate(catalogLine(right)) || left.path.localeCompare(right.path), (left, right) =>
estimate(catalogLine(left)) - estimate(catalogLine(right)) || left.path.localeCompare(right.path),
), ),
})) }))
let used = 0 let used = 0
@ -472,7 +506,9 @@ export const discoveryPlan = <R>(
"- A result typed `Promise<unknown>` has no guaranteed shape — verify what actually came back before relying on its fields.", "- A result typed `Promise<unknown>` has no guaranteed shape — verify what actually came back before relying on its fields.",
"- Run independent calls in parallel: `await Promise.all(items.map((item) => tools.<namespace>.<tool>(item)))`.", "- Run independent calls in parallel: `await Promise.all(items.map((item) => tools.<namespace>.<tool>(item)))`.",
"- `Object.keys(tools)` lists namespaces; `Object.keys(tools.<namespace>)` lists its tools; `for...in` works on both.", "- `Object.keys(tools)` lists namespaces; `Object.keys(tools.<namespace>)` lists its tools; `for...in` works on both.",
...(complete ? [] : ['- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.']), ...(complete
? []
: ['- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.']),
] ]
const syntax = [ const syntax = [
@ -500,26 +536,21 @@ export const discoveryPlan = <R>(
const count = `${group.length} tool${group.length === 1 ? "" : "s"}` const count = `${group.length} tool${group.length === 1 ? "" : "s"}`
// Annotate only when a namespace is not fully shown, so a comprehensive // Annotate only when a namespace is not fully shown, so a comprehensive
// namespace reads cleanly and a truncated one is unambiguous. // namespace reads cleanly and a truncated one is unambiguous.
const label = picked.size === group.length ? count : picked.size === 0 ? `${count}, none shown` : `${count}, ${picked.size} shown` const label =
picked.size === group.length
? count
: picked.size === 0
? `${count}, none shown`
: `${count}, ${picked.size} shown`
toolSection.push(`- ${namespace} (${label})`) toolSection.push(`- ${namespace} (${label})`)
for (const tool of group) if (picked.has(tool)) toolSection.push(catalogLine(tool)) for (const tool of group) if (picked.has(tool)) toolSection.push(catalogLine(tool))
} }
if (!complete) { if (!complete) {
toolSection.push( toolSection.push("", "Search returns complete callable signatures:", `- ${searchSignature}`)
"",
"Search returns complete callable signatures:",
`- ${searchSignature}`,
)
} }
} }
const lines = [ const lines = [...intro, ...workflow, ...rules, ...syntax, ...toolSection]
...intro,
...workflow,
...rules,
...syntax,
...toolSection,
]
return { return {
catalog: described, catalog: described,
instructions: lines.join("\n"), instructions: lines.join("\n"),
@ -534,18 +565,29 @@ export const discoveryPlan = <R>(
* function in JS). An unknown path is an `UnknownTool` error pointing at the working * function in JS). An unknown path is an `UnknownTool` error pointing at the working
* discovery idioms, mirroring how calling an unknown tool fails. * discovery idioms, mirroring how calling an unknown tool fails.
*/ */
const namespaceKeys = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, searchEnabled: boolean): ReadonlyArray<string> => { const namespaceKeys = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string>,
searchEnabled: boolean,
): ReadonlyArray<string> => {
// The reserved discovery namespace is virtual (never present in the host tree); enumerate // The reserved discovery namespace is virtual (never present in the host tree); enumerate
// it explicitly so `Object.keys(tools.$codemode)` matches the callable surface. // it explicitly so `Object.keys(tools.$codemode)` matches the callable surface.
if (searchEnabled && path.length === 1 && path[0] === reservedNamespace) return ["search"] if (searchEnabled && path.length === 1 && path[0] === reservedNamespace) return ["search"]
let value: HostTool<R> | Definition<R> | HostTools<R> = tools let value: HostTool<R> | Definition<R> | HostTools<R> = tools
for (const segment of path) { for (const segment of path) {
if (isBlockedMember(segment) || typeof value === "function" || isDefinition(value) || !Object.hasOwn(value, segment)) { if (
isBlockedMember(segment) ||
typeof value === "function" ||
isDefinition(value) ||
!Object.hasOwn(value, segment)
) {
throw new ToolRuntimeError( throw new ToolRuntimeError(
"UnknownTool", "UnknownTool",
`Unknown tool namespace '${path.join(".")}'.`, `Unknown tool namespace '${path.join(".")}'.`,
searchEnabled searchEnabled
? ["Object.keys(tools) lists the available namespaces; tools.$codemode.search({ query }) finds described tools."] ? [
"Object.keys(tools) lists the available namespaces; tools.$codemode.search({ query }) finds described tools.",
]
: ["Object.keys(tools) lists the available namespaces."], : ["Object.keys(tools) lists the available namespaces."],
) )
} }
@ -555,12 +597,25 @@ const namespaceKeys = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, sear
return Object.keys(value) return Object.keys(value)
} }
const resolve = <R>(tools: HostTools<R>, path: ReadonlyArray<string>, searchEnabled: boolean): HostTool<R> | Definition<R> => { const resolve = <R>(
tools: HostTools<R>,
path: ReadonlyArray<string>,
searchEnabled: boolean,
): HostTool<R> | Definition<R> => {
let value: HostTool<R> | Definition<R> | HostTools<R> = tools let value: HostTool<R> | Definition<R> | HostTools<R> = tools
for (const segment of path) { for (const segment of path) {
if (isBlockedMember(segment) || typeof value === "function" || isDefinition(value) || !Object.hasOwn(value, segment)) { if (
throw new ToolRuntimeError("UnknownTool", `Unknown tool '${path.join(".")}'.`, searchEnabled ? ["Use tools.$codemode.search({ query }) to find available described tools."] : []) isBlockedMember(segment) ||
typeof value === "function" ||
isDefinition(value) ||
!Object.hasOwn(value, segment)
) {
throw new ToolRuntimeError(
"UnknownTool",
`Unknown tool '${path.join(".")}'.`,
searchEnabled ? ["Use tools.$codemode.search({ query }) to find available described tools."] : [],
)
} }
value = value[segment] as HostTool<R> | Definition<R> | HostTools<R> value = value[segment] as HostTool<R> | Definition<R> | HostTools<R>
} }
@ -602,7 +657,8 @@ export const make = <R>(
return effect.pipe( return effect.pipe(
Effect.tap(() => onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "success" })), Effect.tap(() => onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "success" })),
Effect.tapError((error) => Effect.tapError((error) =>
onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "failure", message: failureMessage(error) })), onEnd({ ...call, durationMs: Date.now() - startedAt, outcome: "failure", message: failureMessage(error) }),
),
) )
} }
@ -624,7 +680,7 @@ export const make = <R>(
calls, calls,
keys: (path) => namespaceKeys(tools, path, searchEnabled), keys: (path) => namespaceKeys(tools, path, searchEnabled),
invoke: (path, args) => invoke: (path, args) =>
Effect.gen(function*() { Effect.gen(function* () {
const name = path.join(".") const name = path.join(".")
const externalArgs = args.map((arg) => copyOut(copyIn(arg, `Arguments for tool '${name}'`))) const externalArgs = args.map((arg) => copyOut(copyIn(arg, `Arguments for tool '${name}'`)))
const call = { name } const call = { name }
@ -637,17 +693,32 @@ export const make = <R>(
if (!searchEnabled) throw new ToolRuntimeError("UnknownTool", `Unknown tool '${name}'.`) if (!searchEnabled) throw new ToolRuntimeError("UnknownTool", `Unknown tool '${name}'.`)
const input = externalArgs[0] const input = externalArgs[0]
if (externalArgs.length !== 1 || input === null || typeof input !== "object" || Array.isArray(input)) { if (externalArgs.length !== 1 || input === null || typeof input !== "object" || Array.isArray(input)) {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search expects { query?: string; namespace?: string; limit?: number }.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search expects { query?: string; namespace?: string; limit?: number }.",
)
} }
const request = input as { query?: unknown; namespace?: unknown; limit?: unknown } const request = input as { query?: unknown; namespace?: unknown; limit?: unknown }
if (request.query !== undefined && typeof request.query !== "string") { if (request.query !== undefined && typeof request.query !== "string") {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search query must be a string when provided.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search query must be a string when provided.",
)
} }
if (request.namespace !== undefined && typeof request.namespace !== "string") { if (request.namespace !== undefined && typeof request.namespace !== "string") {
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search namespace must be a string when provided.") throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search namespace must be a string when provided.",
)
} }
if (request.limit !== undefined && (typeof request.limit !== "number" || !Number.isSafeInteger(request.limit) || request.limit <= 0)) { if (
throw new ToolRuntimeError("InvalidToolInput", "tools.$codemode.search limit must be a positive safe integer when provided.") request.limit !== undefined &&
(typeof request.limit !== "number" || !Number.isSafeInteger(request.limit) || request.limit <= 0)
) {
throw new ToolRuntimeError(
"InvalidToolInput",
"tools.$codemode.search limit must be a positive safe integer when provided.",
)
} }
const query = typeof request.query === "string" ? request.query : "" const query = typeof request.query === "string" ? request.query : ""
const namespace = typeof request.namespace === "string" ? request.namespace : undefined const namespace = typeof request.namespace === "string" ? request.namespace : undefined
@ -656,40 +727,50 @@ export const make = <R>(
Effect.try({ Effect.try({
try: () => { try: () => {
const limit = typeof request.limit === "number" ? request.limit : defaultSearchLimit const limit = typeof request.limit === "number" ? request.limit : defaultSearchLimit
const scoped = namespace === undefined ? searchIndex : searchIndex.filter((entry) => entry.namespace === namespace) const scoped =
namespace === undefined ? searchIndex : searchIndex.filter((entry) => entry.namespace === namespace)
// A query that names one tool path exactly (canonical path or rendered // A query that names one tool path exactly (canonical path or rendered
// JavaScript expression) is a lookup, not a search: return that tool alone. // JavaScript expression) is a lookup, not a search: return that tool alone.
const trimmed = query.trim() const trimmed = query.trim()
const pathQuery = trimmed.startsWith("tools.") ? trimmed.slice("tools.".length) : trimmed const pathQuery = trimmed.startsWith("tools.") ? trimmed.slice("tools.".length) : trimmed
const exact = pathQuery === "" ? undefined : scoped.find((entry) => const exact =
entry.description.path === pathQuery || toolExpression(entry.description.path) === trimmed) pathQuery === ""
? undefined
: scoped.find(
(entry) =>
entry.description.path === pathQuery || toolExpression(entry.description.path) === trimmed,
)
const terms = tokenize(query).map(termForms) const terms = tokenize(query).map(termForms)
// Additive field-weighted scoring, summed across terms: exact path or path // Additive field-weighted scoring, summed across terms: exact path or path
// segment (20) > path substring (8) > description substring (4) > any // segment (20) > path substring (8) > description substring (4) > any
// searchable text, incl. input parameter names/descriptions (2). Each term // searchable text, incl. input parameter names/descriptions (2). Each term
// matches a field when any of its forms (the term or a singular variant) // matches a field when any of its forms (the term or a singular variant)
// does. An empty query browses everything, alphabetical by path. // does. An empty query browses everything, alphabetical by path.
const ranked = exact !== undefined const ranked =
? [exact] exact !== undefined
: scoped ? [exact]
.map((entry) => { : scoped
const path = entry.description.path.toLowerCase() .map((entry) => {
const description = entry.description.description.toLowerCase() const path = entry.description.path.toLowerCase()
const score = terms.reduce( const description = entry.description.description.toLowerCase()
(total, forms) => const score = terms.reduce(
total + (total, forms) =>
(forms.some((form) => path === form || path.endsWith(`.${form}`)) ? 20 : 0) + total +
(forms.some((form) => path.includes(form)) ? 8 : 0) + (forms.some((form) => path === form || path.endsWith(`.${form}`)) ? 20 : 0) +
(forms.some((form) => description.includes(form)) ? 4 : 0) + (forms.some((form) => path.includes(form)) ? 8 : 0) +
(forms.some((form) => entry.searchText.includes(form)) ? 2 : 0), (forms.some((form) => description.includes(form)) ? 4 : 0) +
0, (forms.some((form) => entry.searchText.includes(form)) ? 2 : 0),
0,
)
return { entry, score }
})
.filter(({ score }) => terms.length === 0 || score > 0)
.sort(
(left, right) =>
right.score - left.score ||
left.entry.description.path.localeCompare(right.entry.description.path),
) )
return { entry, score } .map(({ entry }) => entry)
})
.filter(({ score }) => terms.length === 0 || score > 0)
.sort((left, right) =>
right.score - left.score || left.entry.description.path.localeCompare(right.entry.description.path))
.map(({ entry }) => entry)
// Result paths are rendered as JavaScript expressions so each `path` is // Result paths are rendered as JavaScript expressions so each `path` is
// directly usable as the call site (`await tools.github.list({ ... })` or // directly usable as the call site (`await tools.github.list({ ... })` or
// `await tools.ns["dashed-name"]({ ... })`). The signature is the pretty, // `await tools.ns["dashed-name"]({ ... })`). The signature is the pretty,
@ -711,10 +792,12 @@ export const make = <R>(
const tool = resolve(tools, path, searchEnabled) const tool = resolve(tools, path, searchEnabled)
let describedInput: unknown let describedInput: unknown
if (isDefinition(tool)) { if (isDefinition(tool)) {
if (externalArgs.length !== 1) throw new ToolRuntimeError("InvalidToolInput", `Tool '${name}' expects exactly one input object.`) if (externalArgs.length !== 1)
throw new ToolRuntimeError("InvalidToolInput", `Tool '${name}' expects exactly one input object.`)
describedInput = yield* Effect.try({ describedInput = yield* Effect.try({
try: () => decodeToolInput(tool, externalArgs[0]), try: () => decodeToolInput(tool, externalArgs[0]),
catch: (cause) => new ToolRuntimeError("InvalidToolInput", `Invalid input for tool '${name}': ${String(cause)}`), catch: (cause) =>
new ToolRuntimeError("InvalidToolInput", `Invalid input for tool '${name}': ${String(cause)}`),
}) })
} }
const input = isDefinition(tool) ? describedInput : externalArgs const input = isDefinition(tool) ? describedInput : externalArgs
@ -722,7 +805,7 @@ export const make = <R>(
const currentCall = { index, name, input } const currentCall = { index, name, input }
if (isDefinition(tool)) { if (isDefinition(tool)) {
return yield* observeEnd( return yield* observeEnd(
Effect.gen(function*() { Effect.gen(function* () {
const raw = yield* runHost(Effect.suspend(() => tool.run(describedInput))) const raw = yield* runHost(Effect.suspend(() => tool.run(describedInput)))
const result = yield* Effect.try({ const result = yield* Effect.try({
try: () => decodeToolOutput(tool, raw), try: () => decodeToolOutput(tool, raw),
@ -734,7 +817,7 @@ export const make = <R>(
) )
} }
return yield* observeEnd( return yield* observeEnd(
Effect.gen(function*() { Effect.gen(function* () {
return yield* decodeOutput(yield* runHost(Effect.suspend(() => tool(...externalArgs))), name) return yield* decodeOutput(yield* runHost(Effect.suspend(() => tool(...externalArgs))), name)
}), }),
currentCall, currentCall,

View file

@ -57,8 +57,7 @@ export type Options<I extends ToolSchema, O extends ToolSchema | undefined, R =
export const isDefinition = <R = never>(value: unknown): value is Definition<R> => export const isDefinition = <R = never>(value: unknown): value is Definition<R> =>
typeof value === "object" && value !== null && "_tag" in value && value._tag === "CodeModeTool" typeof value === "object" && value !== null && "_tag" in value && value._tag === "CodeModeTool"
const isEffectSchema = (schema: ToolSchema): schema is Schema.Decoder<unknown> & Schema.Top => const isEffectSchema = (schema: ToolSchema): schema is Schema.Decoder<unknown> & Schema.Top => Schema.isSchema(schema)
Schema.isSchema(schema)
const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unknown" const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unknown"
@ -69,10 +68,12 @@ const renderLiteral = (value: unknown): string => JSON.stringify(value) ?? "unkn
export const identifierSegment = /^[A-Za-z_$][A-Za-z0-9_$]*$/ export const identifierSegment = /^[A-Za-z_$][A-Za-z0-9_$]*$/
/** Renders a property name as a valid TS object key: bare when an identifier, quoted otherwise. */ /** Renders a property name as a valid TS object key: bare when an identifier, quoted otherwise. */
const renderKey = (name: string): string => identifierSegment.test(name) ? name : JSON.stringify(name) const renderKey = (name: string): string => (identifierSegment.test(name) ? name : JSON.stringify(name))
const effectNumberSentinel = (schema: JsonSchema) => const effectNumberSentinel = (schema: JsonSchema) =>
schema.type === "string" && Array.isArray(schema.enum) && schema.enum.length === 1 && schema.type === "string" &&
Array.isArray(schema.enum) &&
schema.enum.length === 1 &&
(schema.enum[0] === "NaN" || schema.enum[0] === "Infinity" || schema.enum[0] === "-Infinity") (schema.enum[0] === "NaN" || schema.enum[0] === "Infinity" || schema.enum[0] === "-Infinity")
/** /**
@ -118,7 +119,8 @@ const docTags = (schema: JsonSchema): Array<string> => {
*/ */
const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad: string): string => { const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad: string): string => {
const lines = [...(description === undefined ? [] : description.split("\n")), ...tags].map((line) => const lines = [...(description === undefined ? [] : description.split("\n")), ...tags].map((line) =>
line.replaceAll("*/", "* /").replace(/\s+$/, "")) line.replaceAll("*/", "* /").replace(/\s+$/, ""),
)
while (lines.length > 0 && lines[0]!.trim() === "") lines.shift() while (lines.length > 0 && lines[0]!.trim() === "") lines.shift()
while (lines.length > 0 && lines[lines.length - 1]!.trim() === "") lines.pop() while (lines.length > 0 && lines[lines.length - 1]!.trim() === "") lines.pop()
if (lines.length === 0) return "" if (lines.length === 0) return ""
@ -127,7 +129,12 @@ const jsdoc = (description: string | undefined, tags: ReadonlyArray<string>, pad
return `${pad}/**\n${body}\n${pad} */\n` return `${pad}/**\n${body}\n${pad} */\n`
} }
const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: ReadonlySet<string> = new Set()): string => { const renderSchema = (
schema: JsonSchema,
ctx: RenderContext,
depth = 0,
seen: ReadonlySet<string> = new Set(),
): string => {
if (depth > MAX_RENDER_DEPTH) return "unknown" if (depth > MAX_RENDER_DEPTH) return "unknown"
if (schema.$ref) { if (schema.$ref) {
const name = schema.$ref.split("/").pop() const name = schema.$ref.split("/").pop()
@ -146,13 +153,16 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
if ( if (
alternatives.some((item) => item.type === "number") && alternatives.some((item) => item.type === "number") &&
alternatives.every((item) => item.type === "number" || effectNumberSentinel(item)) alternatives.every((item) => item.type === "number" || effectNumberSentinel(item))
) return "number" )
return "number"
// An empty Schema.Struct({}) emits `anyOf: [{ type: "object" }, { type: "array" }]` // An empty Schema.Struct({}) emits `anyOf: [{ type: "object" }, { type: "array" }]`
// (no properties/items); render the bare shape as {} instead of `{} | Array<unknown>`. // (no properties/items); render the bare shape as {} instead of `{} | Array<unknown>`.
if ( if (
alternatives.length === 2 && alternatives.length === 2 &&
alternatives[0]?.type === "object" && alternatives[0].properties === undefined && alternatives[0]?.type === "object" &&
alternatives[1]?.type === "array" && alternatives[1].items === undefined alternatives[0].properties === undefined &&
alternatives[1]?.type === "array" &&
alternatives[1].items === undefined
) { ) {
return "{}" return "{}"
} }
@ -170,7 +180,8 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
const required = new Set(schema.required ?? []) const required = new Set(schema.required ?? [])
const properties = Object.entries(schema.properties ?? {}) const properties = Object.entries(schema.properties ?? {})
const additional = schema.additionalProperties const additional = schema.additionalProperties
const indexType = additional && typeof additional === "object" ? renderSchema(additional, ctx, depth + 1, seen) : undefined const indexType =
additional && typeof additional === "object" ? renderSchema(additional, ctx, depth + 1, seen) : undefined
const field = ([name, value]: readonly [string, JsonSchema]) => const field = ([name, value]: readonly [string, JsonSchema]) =>
`${renderKey(name)}${required.has(name) ? "" : "?"}: ${renderSchema(value, ctx, depth + 1, seen)}` `${renderKey(name)}${required.has(name) ? "" : "?"}: ${renderSchema(value, ctx, depth + 1, seen)}`
@ -183,7 +194,9 @@ const renderSchema = (schema: JsonSchema, ctx: RenderContext, depth = 0, seen: R
// Pretty: an indented block, each described field preceded by its JSDoc comment. // Pretty: an indented block, each described field preceded by its JSDoc comment.
if (properties.length === 0 && indexType === undefined) return "{}" if (properties.length === 0 && indexType === undefined) return "{}"
const pad = " ".repeat(depth + 1) const pad = " ".repeat(depth + 1)
const lines = properties.map((entry) => `${jsdoc(entry[1].description, docTags(entry[1]), pad)}${pad}${field(entry)}`) const lines = properties.map(
(entry) => `${jsdoc(entry[1].description, docTags(entry[1]), pad)}${pad}${field(entry)}`,
)
if (indexType !== undefined) lines.push(`${pad}[key: string]: ${indexType}`) if (indexType !== undefined) lines.push(`${pad}[key: string]: ${indexType}`)
return `{\n${lines.join("\n")}\n${" ".repeat(depth)}}` return `{\n${lines.join("\n")}\n${" ".repeat(depth)}}`
} }
@ -262,7 +275,9 @@ export const inputProperties = <R>(definition: Definition<R>): Array<InputProper
* fields; the default stays the compact single-line form. * fields; the default stays the compact single-line form.
*/ */
export const inputTypeScript = <R>(definition: Definition<R>, pretty = false): string => export const inputTypeScript = <R>(definition: Definition<R>, pretty = false): string =>
isEffectSchema(definition.input) ? toTypeScript(definition.input, false, pretty) : jsonSchemaToTypeScript(definition.input, pretty) isEffectSchema(definition.input)
? toTypeScript(definition.input, false, pretty)
: jsonSchemaToTypeScript(definition.input, pretty)
/** /**
* The model-visible TypeScript type of a tool's result; tools without an output schema * The model-visible TypeScript type of a tool's result; tools without an output schema

View file

@ -49,4 +49,7 @@ export class SandboxSet {
} }
export const isSandboxValue = (value: unknown): value is SandboxDate | SandboxRegExp | SandboxMap | SandboxSet => export const isSandboxValue = (value: unknown): value is SandboxDate | SandboxRegExp | SandboxMap | SandboxSet =>
value instanceof SandboxDate || value instanceof SandboxRegExp || value instanceof SandboxMap || value instanceof SandboxSet value instanceof SandboxDate ||
value instanceof SandboxRegExp ||
value instanceof SandboxMap ||
value instanceof SandboxSet

View file

@ -1,6 +1,13 @@
import { describe, expect, test } from "bun:test" import { describe, expect, test } from "bun:test"
import { Cause, Effect, Schema } from "effect" import { Cause, Effect, Schema } from "effect"
import { CodeMode, ExecuteInputSchema, ExecuteResultSchema, Tool, toolError, type ExecutionLimits } from "../src/index.js" import {
CodeMode,
ExecuteInputSchema,
ExecuteResultSchema,
Tool,
toolError,
type ExecutionLimits,
} from "../src/index.js"
import type { Definition } from "../src/tool.js" import type { Definition } from "../src/tool.js"
const run = (tool: Definition<never>) => const run = (tool: Definition<never>) =>
@ -75,11 +82,14 @@ describe("CodeMode host failure boundary", () => {
output: Schema.Unknown, output: Schema.Unknown,
run: () => run: () =>
Effect.succeed( Effect.succeed(
new Proxy({}, { new Proxy(
ownKeys: () => { {},
throw new Error("host-output-secret") {
ownKeys: () => {
throw new Error("host-output-secret")
},
}, },
}), ),
), ),
}), }),
) )
@ -162,9 +172,7 @@ describe("CodeMode tool-call observation", () => {
) )
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(calls).toStrictEqual([ expect(calls).toStrictEqual([{ index: 0, name: "context.lookup", input: { query: "deployment failure" } }])
{ index: 0, name: "context.lookup", input: { query: "deployment failure" } },
])
}) })
test("observes settled calls with outcome and duration", async () => { test("observes settled calls with outcome and duration", async () => {
@ -173,25 +181,26 @@ describe("CodeMode tool-call observation", () => {
description: "Look up a value", description: "Look up a value",
input: Schema.Struct({ query: Schema.String }), input: Schema.Struct({ query: Schema.String }),
output: Schema.String, output: Schema.String,
run: ({ query }) => run: ({ query }) => (query === "boom" ? Effect.fail(toolError("Lookup refused")) : Effect.succeed(query)),
query === "boom" ? Effect.fail(toolError("Lookup refused")) : Effect.succeed(query),
}) })
const runtime = CodeMode.make({ const runtime = CodeMode.make({
tools: { context: { lookup } }, tools: { context: { lookup } },
onToolCallStart: (call) => Effect.sync(() => { onToolCallStart: (call) =>
events.push({ phase: "start", index: call.index, name: call.name }) Effect.sync(() => {
}), events.push({ phase: "start", index: call.index, name: call.name })
onToolCallEnd: (call) => Effect.sync(() => { }),
expect(call.durationMs).toBeGreaterThanOrEqual(0) onToolCallEnd: (call) =>
events.push({ Effect.sync(() => {
phase: "end", expect(call.durationMs).toBeGreaterThanOrEqual(0)
index: call.index, events.push({
name: call.name, phase: "end",
outcome: call.outcome, index: call.index,
...(call.message === undefined ? {} : { message: call.message }), name: call.name,
}) outcome: call.outcome,
}), ...(call.message === undefined ? {} : { message: call.message }),
})
}),
}) })
const success = await Effect.runPromise(runtime.execute(`return await tools.context.lookup({ query: "ok" })`)) const success = await Effect.runPromise(runtime.execute(`return await tools.context.lookup({ query: "ok" })`))
@ -210,13 +219,15 @@ describe("CodeMode tool-call observation", () => {
describe("CodeMode console capture", () => { describe("CodeMode console capture", () => {
test("captures console output as bounded result logs", async () => { test("captures console output as bounded result logs", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
const returned = console.log("Thread info:", { name: "Demo", count: 2 }) const returned = console.log("Thread info:", { name: "Demo", count: 2 })
console.warn("careful") console.warn("careful")
return returned return returned
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
@ -228,39 +239,45 @@ describe("CodeMode console capture", () => {
}) })
test("keeps logs captured before failures", async () => { test("keeps logs captured before failures", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.log("before failure") console.log("before failure")
throw new Error("boom") throw new Error("boom")
`, `,
})) }),
)
expect(result.ok ? undefined : result.logs).toStrictEqual(["before failure"]) expect(result.ok ? undefined : result.logs).toStrictEqual(["before failure"])
expect(result.ok ? undefined : result.error.message).toBe("Uncaught: boom") expect(result.ok ? undefined : result.error.message).toBe("Uncaught: boom")
}) })
test("prints NaN and Infinity literally instead of the JSON null", async () => { test("prints NaN and Infinity literally instead of the JSON null", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.log(NaN) console.log(NaN)
console.log(Infinity, -Infinity) console.log(Infinity, -Infinity)
console.log({ ratio: NaN, bounds: [Infinity] }) console.log({ ratio: NaN, bounds: [Infinity] })
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual(["NaN", "Infinity -Infinity", '{"ratio":NaN,"bounds":[Infinity]}']) expect(result.logs).toStrictEqual(["NaN", "Infinity -Infinity", '{"ratio":NaN,"bounds":[Infinity]}'])
}) })
test("renders sandbox values nested inside logged containers", async () => { test("renders sandbox values nested inside logged containers", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.log({ m: new Map([["a", 1]]), when: new Date(0), r: /ab/g, s: new Set([1, 2]) }) console.log({ m: new Map([["a", 1]]), when: new Date(0), r: /ab/g, s: new Set([1, 2]) })
console.log([new Date(0)]) console.log([new Date(0)])
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual([ expect(result.logs).toStrictEqual([
@ -270,38 +287,40 @@ describe("CodeMode console capture", () => {
}) })
test("console formatting is total: cycles and opaque references render as markers", async () => { test("console formatting is total: cycles and opaque references render as markers", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
const m = new Map() const m = new Map()
m.set("self", m) m.set("self", m)
console.log({ box: m }) console.log({ box: m })
console.log({ fn: (x) => x, ok: 1 }) console.log({ fn: (x) => x, ok: 1 })
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual([ expect(result.logs).toStrictEqual(['{"box":Map(1) [["self",[Circular]]]}', '{"fn":[CodeMode reference],"ok":1}'])
'{"box":Map(1) [["self",[Circular]]]}',
'{"fn":[CodeMode reference],"ok":1}',
])
}) })
test("console.table renders sandbox value cells", async () => { test("console.table renders sandbox value cells", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.table([{ when: new Date(0), n: NaN }]) console.table([{ when: new Date(0), n: NaN }])
return null return null
`, `,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(result.logs).toStrictEqual(["(index)\twhen\tn\n0\t1970-01-01T00:00:00.000Z\tNaN"]) expect(result.logs).toStrictEqual(["(index)\twhen\tn\n0\t1970-01-01T00:00:00.000Z\tNaN"])
}) })
test("captures console.dir and console.table output", async () => { test("captures console.dir and console.table output", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.dir({ nested: { ok: true } }) console.dir({ nested: { ok: true } })
console.table([ console.table([
{ name: "Kit", count: 1, hidden: "x" }, { name: "Kit", count: 1, hidden: "x" },
@ -309,15 +328,13 @@ describe("CodeMode console capture", () => {
], ["name", "count"]) ], ["name", "count"])
return "done" return "done"
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: "done", value: "done",
logs: [ logs: ['{"nested":{"ok":true}}', "(index)\tname\tcount\n0\tKit\t1\n1\tOlive\t2"],
'{"nested":{"ok":true}}',
"(index)\tname\tcount\n0\tKit\t1\n1\tOlive\t2",
],
toolCalls: [], toolCalls: [],
}) })
}) })
@ -325,9 +342,11 @@ describe("CodeMode console capture", () => {
describe("CodeMode output budget", () => { describe("CodeMode output budget", () => {
test("absent maxOutputBytes means no truncation at all", async () => { test("absent maxOutputBytes means no truncation at all", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: `console.log("z".repeat(50_000)); return "x".repeat(100_000)`, CodeMode.execute({
})) code: `console.log("z".repeat(50_000)); return "x".repeat(100_000)`,
}),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
@ -338,29 +357,35 @@ describe("CodeMode output budget", () => {
test("truncates an oversized result value with a marker instead of failing", async () => { test("truncates an oversized result value with a marker instead of failing", async () => {
const limits: ExecutionLimits = { maxOutputBytes: 40 } const limits: ExecutionLimits = { maxOutputBytes: 40 }
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: `return { data: "${"x".repeat(200)}" }`, CodeMode.execute({
limits, code: `return { data: "${"x".repeat(200)}" }`,
})) limits,
}),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
expect(result.truncated).toBe(true) expect(result.truncated).toBe(true)
expect(typeof result.value).toBe("string") expect(typeof result.value).toBe("string")
expect(result.value).toMatch(/^\{"data":"x+ \[result truncated: \d+ bytes exceeds the 40-byte output limit; return a smaller value\]$/) expect(result.value).toMatch(
/^\{"data":"x+ \[result truncated: \d+ bytes exceeds the 40-byte output limit; return a smaller value\]$/,
)
expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(result)))).toStrictEqual(result) expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(result)))).toStrictEqual(result)
}) })
test("keeps leading logs within the remaining budget and marks the cut", async () => { test("keeps leading logs within the remaining budget and marks the cut", async () => {
const limits: ExecutionLimits = { maxOutputBytes: 40 } const limits: ExecutionLimits = { maxOutputBytes: 40 }
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.log("first line") console.log("first line")
console.log("${"y".repeat(200)}") console.log("${"y".repeat(200)}")
return "ok" return "ok"
`, `,
limits, limits,
})) }),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
@ -370,12 +395,14 @@ describe("CodeMode output budget", () => {
}) })
test("does not mark results within the budget", async () => { test("does not mark results within the budget", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: ` CodeMode.execute({
code: `
console.log("fits") console.log("fits")
return { fits: true } return { fits: true }
`, `,
})) }),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: { fits: true }, value: { fits: true },
@ -395,18 +422,21 @@ describe("CodeMode schema flexibility", () => {
properties: { id: { type: "string" }, count: { type: "number" } }, properties: { id: { type: "string" }, count: { type: "number" } },
required: ["id"], required: ["id"],
}, },
run: (input) => Effect.sync(() => { run: (input) =>
observed.push(input) Effect.sync(() => {
return { echoed: input } observed.push(input)
}), return { echoed: input }
}),
}) })
const runtime = CodeMode.make({ tools: { adapter: { call } } }) const runtime = CodeMode.make({ tools: { adapter: { call } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
path: "adapter.call", {
description: "Call an adapter-described tool", path: "adapter.call",
signature: "tools.adapter.call(input: { id: string; count?: number }): Promise<unknown>", description: "Call an adapter-described tool",
}]) signature: "tools.adapter.call(input: { id: string; count?: number }): Promise<unknown>",
},
])
// JSON Schema is render-only: mistyped input passes through unvalidated. // JSON Schema is render-only: mistyped input passes through unvalidated.
const result = await Effect.runPromise(runtime.execute(`return await tools.adapter.call({ id: 42 })`)) const result = await Effect.runPromise(runtime.execute(`return await tools.adapter.call({ id: 42 })`))
@ -421,17 +451,25 @@ describe("CodeMode schema flexibility", () => {
input: { type: "object", properties: { login: { type: "string" } }, required: ["login"] }, input: { type: "object", properties: { login: { type: "string" } }, required: ["login"] },
output: { output: {
$ref: "#/$defs/User", $ref: "#/$defs/User",
$defs: { User: { type: "object", properties: { login: { type: "string" }, id: { type: "number" } }, required: ["login", "id"] } }, $defs: {
User: {
type: "object",
properties: { login: { type: "string" }, id: { type: "number" } },
required: ["login", "id"],
},
},
}, },
run: () => Effect.succeed({ login: "kit", id: 7 }), run: () => Effect.succeed({ login: "kit", id: 7 }),
}) })
const runtime = CodeMode.make({ tools: { users: { lookup } } }) const runtime = CodeMode.make({ tools: { users: { lookup } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
path: "users.lookup", {
description: "Look up a user", path: "users.lookup",
signature: "tools.users.lookup(input: { login: string }): Promise<{ login: string; id: number }>", description: "Look up a user",
}]) signature: "tools.users.lookup(input: { login: string }): Promise<{ login: string; id: number }>",
},
])
const result = await Effect.runPromise(runtime.execute(`return await tools.users.lookup({ login: "kit" })`)) const result = await Effect.runPromise(runtime.execute(`return await tools.users.lookup({ login: "kit" })`))
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
@ -478,16 +516,20 @@ describe("CodeMode public contract", () => {
expect(agentTool.input).toBe(ExecuteInputSchema) expect(agentTool.input).toBe(ExecuteInputSchema)
expect(agentTool.output).toBe(ExecuteResultSchema) expect(agentTool.output).toBe(ExecuteResultSchema)
expect(agentTool.description).toBe(runtime.instructions()) expect(agentTool.description).toBe(runtime.instructions())
expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(projected)))).toStrictEqual(projected) expect(Schema.decodeUnknownSync(ExecuteResultSchema)(JSON.parse(JSON.stringify(projected)))).toStrictEqual(
projected,
)
}) })
test("inlines a COMPLETE small catalog and keeps search registered but unadvertised", async () => { test("inlines a COMPLETE small catalog and keeps search registered but unadvertised", async () => {
const runtime = CodeMode.make({ tools }) const runtime = CodeMode.make({ tools })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
path: "orders.lookup", {
description: "Look up an order by ID", path: "orders.lookup",
signature: "tools.orders.lookup(input: { id: string }): Promise<{ id: string; status: string }>", description: "Look up an order by ID",
}]) signature: "tools.orders.lookup(input: { id: string }): Promise<{ id: string; status: string }>",
},
])
expect(runtime.instructions()).toContain("Available tools (COMPLETE list") expect(runtime.instructions()).toContain("Available tools (COMPLETE list")
expect(runtime.instructions()).toContain("- orders (1 tool)") expect(runtime.instructions()).toContain("- orders (1 tool)")
expect(runtime.instructions()).toContain( expect(runtime.instructions()).toContain(
@ -502,11 +544,13 @@ describe("CodeMode public contract", () => {
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (result.ok) { if (result.ok) {
expect(result.value).toStrictEqual({ expect(result.value).toStrictEqual({
items: [{ items: [
path: "tools.orders.lookup", {
description: "Look up an order by ID", path: "tools.orders.lookup",
signature: "tools.orders.lookup(input: {\n id: string\n}): Promise<{\n id: string\n status: string\n}>", description: "Look up an order by ID",
}], signature: "tools.orders.lookup(input: {\n id: string\n}): Promise<{\n id: string\n status: string\n}>",
},
],
total: 1, total: 1,
}) })
} }
@ -521,31 +565,43 @@ describe("CodeMode public contract", () => {
}) })
const runtime = CodeMode.make({ tools: { context7: { "resolve-library-id": resolveLibrary } } }) const runtime = CodeMode.make({ tools: { context7: { "resolve-library-id": resolveLibrary } } })
expect(runtime.catalog()).toStrictEqual([{ expect(runtime.catalog()).toStrictEqual([
path: "context7.resolve-library-id", {
description: "Resolve a library ID", path: "context7.resolve-library-id",
signature: 'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>', description: "Resolve a library ID",
}]) signature: 'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>',
expect(runtime.instructions()).toContain('tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>') },
])
expect(runtime.instructions()).toContain(
'tools.context7["resolve-library-id"](input: { libraryName: string }): Promise<string>',
)
const search = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: "resolve library id" })`)) const search = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: "resolve library id" })`),
)
expect(search.ok).toBe(true) expect(search.ok).toBe(true)
if (search.ok) { if (search.ok) {
expect(search.value).toStrictEqual({ expect(search.value).toStrictEqual({
items: [{ items: [
path: 'tools.context7["resolve-library-id"]', {
description: "Resolve a library ID", path: 'tools.context7["resolve-library-id"]',
signature: 'tools.context7["resolve-library-id"](input: {\n libraryName: string\n}): Promise<string>', description: "Resolve a library ID",
}], signature: 'tools.context7["resolve-library-id"](input: {\n libraryName: string\n}): Promise<string>',
},
],
total: 1, total: 1,
}) })
} }
const call = await Effect.runPromise(runtime.execute(`return await tools.context7["resolve-library-id"]({ libraryName: "TypeScript" })`)) const call = await Effect.runPromise(
runtime.execute(`return await tools.context7["resolve-library-id"]({ libraryName: "TypeScript" })`),
)
expect(call.ok).toBe(true) expect(call.ok).toBe(true)
if (call.ok) expect(call.value).toBe("/resolved/TypeScript") if (call.ok) expect(call.value).toBe("/resolved/TypeScript")
const exact = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: 'tools.context7["resolve-library-id"]' })`)) const exact = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: 'tools.context7["resolve-library-id"]' })`),
)
expect(exact.ok).toBe(true) expect(exact.ok).toBe(true)
if (exact.ok) expect((exact.value as { total: number }).total).toBe(1) if (exact.ok) expect((exact.value as { total: number }).total).toBe(1)
}) })
@ -561,7 +617,9 @@ describe("CodeMode public contract", () => {
expect(instructions.indexOf("## Rules")).toBeLessThan(instructions.indexOf("## Syntax")) expect(instructions.indexOf("## Rules")).toBeLessThan(instructions.indexOf("## Syntax"))
expect(instructions.indexOf("## Syntax")).toBeLessThan(instructions.indexOf("\n## Available tools (COMPLETE list")) expect(instructions.indexOf("## Syntax")).toBeLessThan(instructions.indexOf("\n## Available tools (COMPLETE list"))
// The workflow carries the result-shape guidance; Rules only add content beyond it. // The workflow carries the result-shape guidance; Rules only add content beyond it.
expect(instructions).toContain('`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string') expect(instructions).toContain(
'`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string',
)
expect(instructions).toContain("Return only the fields you need") expect(instructions).toContain("Return only the fields you need")
expect(instructions).toContain("raw payloads get truncated and waste context") expect(instructions).toContain("raw payloads get truncated and waste context")
expect(instructions).toContain("`const res = await tools.<namespace>.<tool>(input)`") expect(instructions).toContain("`const res = await tools.<namespace>.<tool>(input)`")
@ -584,8 +642,12 @@ describe("CodeMode public contract", () => {
expect(partial).toContain( expect(partial).toContain(
'1. Find a tool (skip when it is already listed below): `const { items } = await tools.$codemode.search({ query: "<intent + key nouns>" })` — short phrases like "list issues" work best.', '1. Find a tool (skip when it is already listed below): `const { items } = await tools.$codemode.search({ query: "<intent + key nouns>" })` — short phrases like "list issues" work best.',
) )
expect(partial).toContain("Only tools listed here or returned by `tools.$codemode.search` are available inside `tools`") expect(partial).toContain(
expect(partial).toContain('- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.') "Only tools listed here or returned by `tools.$codemode.search` are available inside `tools`",
)
expect(partial).toContain(
'- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.',
)
expect(partial).not.toContain("total_count") expect(partial).not.toContain("total_count")
expect(partial).not.toContain("tools.orders.lookup({") expect(partial).not.toContain("tools.orders.lookup({")
}) })
@ -604,7 +666,9 @@ describe("CodeMode public contract", () => {
expect(instructions).not.toContain("instanceof Error") expect(instructions).not.toContain("instanceof Error")
expect(instructions).not.toContain("splice") expect(instructions).not.toContain("splice")
// The data-boundary note survives. // The data-boundary note survives.
expect(instructions).toContain("Dates serialize to ISO strings at data boundaries; Map/Set/RegExp serialize to `{}`.") expect(instructions).toContain(
"Dates serialize to ISO strings at data boundaries; Map/Set/RegExp serialize to `{}`.",
)
}) })
test("zero tools keep minimal sections and the no-tools notice", () => { test("zero tools keep minimal sections and the no-tools notice", () => {
@ -635,18 +699,22 @@ describe("CodeMode public contract", () => {
tools: { thread: { uploadFile: upload, generateImage: generate }, orders: { lookup } }, tools: { thread: { uploadFile: upload, generateImage: generate }, orders: { lookup } },
discovery: { maxInlineCatalogTokens: 0 }, discovery: { maxInlineCatalogTokens: 0 },
}) })
expect(runtime.instructions()).toContain("Available tools (PARTIAL — 0 of 3 shown; find the rest with tools.$codemode.search)") expect(runtime.instructions()).toContain(
"Available tools (PARTIAL — 0 of 3 shown; find the rest with tools.$codemode.search)",
)
expect(runtime.instructions()).toContain("- thread (2 tools, none shown)") expect(runtime.instructions()).toContain("- thread (2 tools, none shown)")
expect(runtime.instructions()).toContain("- orders (1 tool, none shown)") expect(runtime.instructions()).toContain("- orders (1 tool, none shown)")
expect(runtime.instructions()).toMatch(/\$codemode\.search/) expect(runtime.instructions()).toMatch(/\$codemode\.search/)
expect(runtime.instructions()).not.toMatch(/tools\.thread\.uploadFile\(input/) expect(runtime.instructions()).not.toMatch(/tools\.thread\.uploadFile\(input/)
const result = await Effect.runPromise(runtime.execute(` const result = await Effect.runPromise(
runtime.execute(`
return await tools.$codemode.search({ return await tools.$codemode.search({
query: "send message attachment upload file to current Discord thread", query: "send message attachment upload file to current Discord thread",
limit: 2 limit: 2
}) })
`)) `),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) return if (!result.ok) return
expect(result.value).toStrictEqual({ expect(result.value).toStrictEqual({
@ -666,19 +734,27 @@ describe("CodeMode public contract", () => {
}) })
expect(result.toolCalls).toStrictEqual([{ name: "$codemode.search" }]) expect(result.toolCalls).toStrictEqual([{ name: "$codemode.search" }])
const variants = await Effect.runPromise(runtime.execute(` const variants = await Effect.runPromise(
runtime.execute(`
return await Promise.all([ return await Promise.all([
tools.$codemode.search({ query: "file" }), tools.$codemode.search({ query: "file" }),
tools.$codemode.search({ query: "image" }) tools.$codemode.search({ query: "image" })
]) ])
`)) `),
)
expect(variants.ok).toBe(true) expect(variants.ok).toBe(true)
if (variants.ok) { if (variants.ok) {
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[0]?.items[0]?.path).toBe("tools.thread.uploadFile") expect((variants.value as Array<{ items: Array<{ path: string }> }>)[0]?.items[0]?.path).toBe(
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[1]?.items[0]?.path).toBe("tools.thread.generateImage") "tools.thread.uploadFile",
)
expect((variants.value as Array<{ items: Array<{ path: string }> }>)[1]?.items[0]?.path).toBe(
"tools.thread.generateImage",
)
} }
const removed = await Effect.runPromise(runtime.execute(`return await tools.$codemode.describe({ path: "thread.uploadFile" })`)) const removed = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.describe({ path: "thread.uploadFile" })`),
)
expect(removed.ok).toBe(false) expect(removed.ok).toBe(false)
if (!removed.ok) expect(removed.error.kind).toBe("UnknownTool") if (!removed.ok) expect(removed.error.kind).toBe("UnknownTool")
}) })
@ -706,15 +782,19 @@ describe("CodeMode public contract", () => {
} }
for (const query of ["many.tool13", "tools.many.tool13"]) { for (const query of ["many.tool13", "tools.many.tool13"]) {
const exact = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`)) const exact = await Effect.runPromise(
runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`),
)
expect(exact.ok).toBe(true) expect(exact.ok).toBe(true)
if (exact.ok) { if (exact.ok) {
expect(exact.value).toStrictEqual({ expect(exact.value).toStrictEqual({
items: [{ items: [
path: "tools.many.tool13", {
description: "Numbered tool 13", path: "tools.many.tool13",
signature: "tools.many.tool13(input: {\n id: string\n}): Promise<string>", description: "Numbered tool 13",
}], signature: "tools.many.tool13(input: {\n id: string\n}): Promise<string>",
},
],
total: 1, total: 1,
}) })
} }
@ -737,20 +817,23 @@ describe("CodeMode public contract", () => {
}) })
// Empty query + namespace browses just that namespace, alphabetical by path. // Empty query + namespace browses just that namespace, alphabetical by path.
const browse = await Effect.runPromise(runtime.execute( const browse = await Effect.runPromise(
`return await tools.$codemode.search({ query: "", namespace: "github" })`, runtime.execute(`return await tools.$codemode.search({ query: "", namespace: "github" })`),
)) )
expect(browse.ok).toBe(true) expect(browse.ok).toBe(true)
if (browse.ok) { if (browse.ok) {
const value = browse.value as { items: Array<{ path: string }>; total: number } const value = browse.value as { items: Array<{ path: string }>; total: number }
expect(value.total).toBe(2) expect(value.total).toBe(2)
expect(value.items.map((item) => item.path)).toStrictEqual(["tools.github.create_issue", "tools.github.list_issues"]) expect(value.items.map((item) => item.path)).toStrictEqual([
"tools.github.create_issue",
"tools.github.list_issues",
])
} }
// A query + namespace ranks within that namespace only. // A query + namespace ranks within that namespace only.
const scoped = await Effect.runPromise(runtime.execute( const scoped = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: "linear" })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: "linear" })`),
)) )
expect(scoped.ok).toBe(true) expect(scoped.ok).toBe(true)
if (scoped.ok) { if (scoped.ok) {
const value = scoped.value as { items: Array<{ path: string }>; total: number } const value = scoped.value as { items: Array<{ path: string }>; total: number }
@ -758,9 +841,9 @@ describe("CodeMode public contract", () => {
expect(value.items[0]?.path).toBe("tools.linear.list_issues") expect(value.items[0]?.path).toBe("tools.linear.list_issues")
} }
const invalid = await Effect.runPromise(runtime.execute( const invalid = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: 7 })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: 7 })`),
)) )
expect(invalid.ok).toBe(false) expect(invalid.ok).toBe(false)
if (!invalid.ok) expect(invalid.error.kind).toBe("InvalidToolInput") if (!invalid.ok) expect(invalid.error.kind).toBe("InvalidToolInput")
}) })
@ -785,9 +868,9 @@ describe("CodeMode public contract", () => {
// "attachment" appears in neither path nor description — only in the input schema's // "attachment" appears in neither path nor description — only in the input schema's
// property names, which the searchable text includes. // property names, which the searchable text includes.
const byParameter = await Effect.runPromise(runtime.execute( const byParameter = await Effect.runPromise(
`return await tools.$codemode.search({ query: "attachment" })`, runtime.execute(`return await tools.$codemode.search({ query: "attachment" })`),
)) )
expect(byParameter.ok).toBe(true) expect(byParameter.ok).toBe(true)
if (byParameter.ok) { if (byParameter.ok) {
const value = byParameter.value as { items: Array<{ path: string }>; total: number } const value = byParameter.value as { items: Array<{ path: string }>; total: number }
@ -796,9 +879,9 @@ describe("CodeMode public contract", () => {
} }
// Substring matching: a partial word ("docum") still hits the description. // Substring matching: a partial word ("docum") still hits the description.
const bySubstring = await Effect.runPromise(runtime.execute( const bySubstring = await Effect.runPromise(
`return await tools.$codemode.search({ query: "docum" })`, runtime.execute(`return await tools.$codemode.search({ query: "docum" })`),
)) )
expect(bySubstring.ok).toBe(true) expect(bySubstring.ok).toBe(true)
if (bySubstring.ok) { if (bySubstring.ok) {
const value = bySubstring.value as { items: Array<{ path: string }>; total: number } const value = bySubstring.value as { items: Array<{ path: string }>; total: number }
@ -825,9 +908,9 @@ describe("CodeMode public contract", () => {
}) })
// "issues" still finds the singular-only tool (term OR singular(term) per field)... // "issues" still finds the singular-only tool (term OR singular(term) per field)...
const plural = await Effect.runPromise(runtime.execute( const plural = await Effect.runPromise(
`return await tools.$codemode.search({ query: "issues", namespace: "tracker" })`, runtime.execute(`return await tools.$codemode.search({ query: "issues", namespace: "tracker" })`),
)) )
expect(plural.ok).toBe(true) expect(plural.ok).toBe(true)
if (plural.ok) { if (plural.ok) {
const value = plural.value as { items: Array<{ path: string }>; total: number } const value = plural.value as { items: Array<{ path: string }>; total: number }
@ -836,14 +919,15 @@ describe("CodeMode public contract", () => {
} }
// ...while a true "issues" path match still outranks the singular-only description match. // ...while a true "issues" path match still outranks the singular-only description match.
const ranked = await Effect.runPromise(runtime.execute( const ranked = await Effect.runPromise(runtime.execute(`return await tools.$codemode.search({ query: "issues" })`))
`return await tools.$codemode.search({ query: "issues" })`,
))
expect(ranked.ok).toBe(true) expect(ranked.ok).toBe(true)
if (ranked.ok) { if (ranked.ok) {
const value = ranked.value as { items: Array<{ path: string }>; total: number } const value = ranked.value as { items: Array<{ path: string }>; total: number }
expect(value.total).toBe(2) expect(value.total).toBe(2)
expect(value.items.map((item) => item.path)).toStrictEqual(["tools.github.list_issues", "tools.tracker.fetch_all"]) expect(value.items.map((item) => item.path)).toStrictEqual([
"tools.github.list_issues",
"tools.tracker.fetch_all",
])
} }
}) })
@ -882,8 +966,12 @@ describe("CodeMode public contract", () => {
run: () => Effect.succeed("ok"), run: () => Effect.succeed("ok"),
}) })
const expensive = Tool.make({ const expensive = Tool.make({
description: "An expensive tool whose description alone consumes far more than the remaining inline catalog byte budget for this runtime", description:
input: Schema.Struct({ someRatherLongParameterName: Schema.String, anotherEvenLongerParameterName: Schema.Number }), "An expensive tool whose description alone consumes far more than the remaining inline catalog byte budget for this runtime",
input: Schema.Struct({
someRatherLongParameterName: Schema.String,
anotherEvenLongerParameterName: Schema.Number,
}),
output: Schema.String, output: Schema.String,
run: () => Effect.succeed("ok"), run: () => Effect.succeed("ok"),
}) })
@ -896,7 +984,9 @@ describe("CodeMode public contract", () => {
}) })
const instructions = runtime.instructions() const instructions = runtime.instructions()
expect(instructions).toContain("Available tools (PARTIAL — 2 of 3 shown; find the rest with tools.$codemode.search)") expect(instructions).toContain(
"Available tools (PARTIAL — 2 of 3 shown; find the rest with tools.$codemode.search)",
)
expect(instructions).toContain("- alpha (2 tools, 1 shown)") expect(instructions).toContain("- alpha (2 tools, 1 shown)")
expect(instructions).toContain(" - tools.alpha.cheap(input: { q: string }): Promise<string> // Cheap") expect(instructions).toContain(" - tools.alpha.cheap(input: { q: string }): Promise<string> // Cheap")
expect(instructions).not.toContain("tools.alpha.expensive(") expect(instructions).not.toContain("tools.alpha.expensive(")
@ -912,10 +1002,11 @@ describe("CodeMode public contract", () => {
description: "Double a number", description: "Double a number",
input: Schema.Struct({ value: Schema.NumberFromString }), input: Schema.Struct({ value: Schema.NumberFromString }),
output: Schema.NumberFromString, output: Schema.NumberFromString,
run: ({ value }) => Effect.sync(() => { run: ({ value }) =>
observed.push(value) Effect.sync(() => {
return String(value * 2) observed.push(value)
}), return String(value * 2)
}),
}) })
const runtime = CodeMode.make({ const runtime = CodeMode.make({
tools: { math: { double: transformed } }, tools: { math: { double: transformed } },
@ -934,9 +1025,11 @@ describe("CodeMode public contract", () => {
}) })
test("returns JSON-safe data and normalizes undefined to null", async () => { test("returns JSON-safe data and normalizes undefined to null", async () => {
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
code: `return { top: undefined, nested: [1, undefined] }`, CodeMode.execute({
})) code: `return { top: undefined, nested: [1, undefined] }`,
}),
)
expect(result).toStrictEqual({ expect(result).toStrictEqual({
ok: true, ok: true,
value: { top: null, nested: [1, null] }, value: { top: null, nested: [1, null] },
@ -947,18 +1040,20 @@ describe("CodeMode public contract", () => {
test("rejects invalid configuration and discovery limits", async () => { test("rejects invalid configuration and discovery limits", async () => {
expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: 0 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: 0 } })).toThrow(RangeError)
expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: Number.POSITIVE_INFINITY } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { timeoutMs: Number.POSITIVE_INFINITY } })).toThrow(
RangeError,
)
expect(() => CodeMode.execute({ code: "return 1", limits: { maxToolCalls: -1 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { maxToolCalls: -1 } })).toThrow(RangeError)
expect(() => CodeMode.execute({ code: "return 1", limits: { maxOutputBytes: -1 } })).toThrow(RangeError) expect(() => CodeMode.execute({ code: "return 1", limits: { maxOutputBytes: -1 } })).toThrow(RangeError)
expect(() => CodeMode.make({ tools, discovery: { maxInlineCatalogTokens: -1 } })).toThrow(RangeError) expect(() => CodeMode.make({ tools, discovery: { maxInlineCatalogTokens: -1 } })).toThrow(RangeError)
const result = await Effect.runPromise(CodeMode.make({ const result = await Effect.runPromise(
tools, CodeMode.make({
discovery: { maxInlineCatalogTokens: 0 }, tools,
}).execute( discovery: { maxInlineCatalogTokens: 0 },
`return await tools.$codemode.search({ query: "order", limit: 0.5 })`, }).execute(`return await tools.$codemode.search({ query: "order", limit: 0.5 })`),
)) )
expect(result.ok).toBe(false) expect(result.ok).toBe(false)
if (result.ok) return if (result.ok) return
expect(result.error.kind).toBe("InvalidToolInput") expect(result.error.kind).toBe("InvalidToolInput")
@ -979,14 +1074,16 @@ describe("CodeMode public contract", () => {
output: Schema.Number, output: Schema.Number,
run: () => Effect.succeed(1), run: () => Effect.succeed(1),
}) })
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
tools: { host: { count: counter } }, CodeMode.execute({
code: ` tools: { host: { count: counter } },
code: `
let total = 0 let total = 0
for (let i = 0; i < 150; i += 1) total += await tools.host.count({}) for (let i = 0; i < 150; i += 1) total += await tools.host.count({})
return total return total
`, `,
})) }),
)
expect(result).toMatchObject({ ok: true, value: 150 }) expect(result).toMatchObject({ ok: true, value: 150 })
if (result.ok) expect(result.toolCalls.length).toBe(150) if (result.ok) expect(result.toolCalls.length).toBe(150)
}) })
@ -1008,8 +1105,6 @@ describe("CodeMode public contract", () => {
}) })
test("reserves the discovery namespace", () => { test("reserves the discovery namespace", () => {
expect(() => CodeMode.make({ tools: { $codemode: { lookup } } })).toThrow( expect(() => CodeMode.make({ tools: { $codemode: { lookup } } })).toThrow(/reserved for CodeMode discovery tools/)
/reserved for CodeMode discovery tools/,
)
}) })
}) })

View file

@ -36,10 +36,12 @@ const error = async (code: string) => {
describe("Object.keys over tool references", () => { describe("Object.keys over tool references", () => {
test("enumerates top-level namespaces (the transcript program)", async () => { test("enumerates top-level namespaces (the transcript program)", async () => {
expect(await value(` expect(
await value(`
const namespaces = Object.keys(tools) const namespaces = Object.keys(tools)
return { namespaces, count: namespaces.length } return { namespaces, count: namespaces.length }
`)).toEqual({ namespaces: ["github", "memory", "playwright"], count: 3 }) `),
).toEqual({ namespaces: ["github", "memory", "playwright"], count: 3 })
}) })
test("enumerates tool names at a nested namespace", async () => { test("enumerates tool names at a nested namespace", async () => {
@ -92,7 +94,8 @@ describe("Object.keys over arrays", () => {
describe("for...in", () => { describe("for...in", () => {
test("iterates own enumerable keys of a plain object with break/continue", async () => { test("iterates own enumerable keys of a plain object with break/continue", async () => {
expect(await value(` expect(
await value(`
const seen = [] const seen = []
for (const key in { a: 1, b: 2, c: 3, d: 4 }) { for (const key in { a: 1, b: 2, c: 3, d: 4 }) {
if (key === "b") continue if (key === "b") continue
@ -100,41 +103,50 @@ describe("for...in", () => {
seen.push(key) seen.push(key)
} }
return seen return seen
`)).toEqual(["a", "c"]) `),
).toEqual(["a", "c"])
}) })
test("iterates index strings over arrays", async () => { test("iterates index strings over arrays", async () => {
expect(await value(` expect(
await value(`
const indexes = [] const indexes = []
for (const i in ["x", "y", "z"]) { for (const i in ["x", "y", "z"]) {
if (i === "2") break if (i === "2") break
indexes.push(i) indexes.push(i)
} }
return indexes return indexes
`)).toEqual(["0", "1"]) `),
).toEqual(["0", "1"])
}) })
test("supports let declarations and bare identifiers", async () => { test("supports let declarations and bare identifiers", async () => {
expect(await value(` expect(
await value(`
let last = "" let last = ""
for (let key in { a: 1, b: 2 }) last = key for (let key in { a: 1, b: 2 }) last = key
return last return last
`)).toBe("b") `),
expect(await value(` ).toBe("b")
expect(
await value(`
let key = "before" let key = "before"
for (key in { only: 1 }) {} for (key in { only: 1 }) {}
return key return key
`)).toBe("only") `),
).toBe("only")
}) })
test("enumerates namespaces and tools from the host tool tree", async () => { test("enumerates namespaces and tools from the host tool tree", async () => {
expect(await value(` expect(
await value(`
const names = [] const names = []
for (const ns in tools) { for (const ns in tools) {
for (const name in tools[ns]) names.push(ns + "." + name) for (const name in tools[ns]) names.push(ns + "." + name)
} }
return names return names
`)).toEqual(["github.list_issues", "github.get_issue", "memory.search", "playwright.navigate"]) `),
).toEqual(["github.list_issues", "github.get_issue", "memory.search", "playwright.navigate"])
}) })
test("unsupported values fail with a hint at for...of and Object.keys", async () => { test("unsupported values fail with a hint at for...of and Object.keys", async () => {

View file

@ -143,21 +143,40 @@ describe("H1: NaN/Infinity flow as intermediates and normalize to null at the bo
describe("Error values and instanceof", () => { describe("Error values and instanceof", () => {
test("new Error carries name/message and is instanceof Error", async () => { test("new Error carries name/message and is instanceof Error", async () => {
expect(await value(`const e = new Error("boom"); return [e instanceof Error, e.name, e.message]`)).toEqual([true, "Error", "boom"]) expect(await value(`const e = new Error("boom"); return [e instanceof Error, e.name, e.message]`)).toEqual([
true,
"Error",
"boom",
])
}) })
test("Error without new behaves like new Error", async () => { test("Error without new behaves like new Error", async () => {
expect(await value(`const e = Error("plain"); return [e instanceof Error, e.name, e.message]`)).toEqual([true, "Error", "plain"]) expect(await value(`const e = Error("plain"); return [e instanceof Error, e.name, e.message]`)).toEqual([
expect(await value(`const e = new Error(); return [e.name, e.message, e instanceof Error]`)).toEqual(["Error", "", true]) true,
"Error",
"plain",
])
expect(await value(`const e = new Error(); return [e.name, e.message, e instanceof Error]`)).toEqual([
"Error",
"",
true,
])
}) })
test("specific error types are instanceof themselves and Error, not each other", async () => { test("specific error types are instanceof themselves and Error, not each other", async () => {
expect(await value(`const e = new TypeError("t"); return [e instanceof TypeError, e instanceof Error, e instanceof RangeError]`)).toEqual([true, true, false]) expect(
await value(
`const e = new TypeError("t"); return [e instanceof TypeError, e instanceof Error, e instanceof RangeError]`,
),
).toEqual([true, true, false])
expect(await value(`return new Error("e") instanceof TypeError`)).toBe(false) expect(await value(`return new Error("e") instanceof TypeError`)).toBe(false)
}) })
test("thrown errors keep instanceof through try/catch", async () => { test("thrown errors keep instanceof through try/catch", async () => {
expect(await value(`try { throw new Error("x") } catch (e) { return [e instanceof Error, e.message] }`)).toEqual([true, "x"]) expect(await value(`try { throw new Error("x") } catch (e) { return [e instanceof Error, e.message] }`)).toEqual([
true,
"x",
])
}) })
test("interpreter runtime failures are caught as Error values", async () => { test("interpreter runtime failures are caught as Error values", async () => {
@ -168,33 +187,48 @@ describe("Error values and instanceof", () => {
test("caught failures carry the constructor name the real-JS failure would have", async () => { test("caught failures carry the constructor name the real-JS failure would have", async () => {
// JSON.parse throws SyntaxError: name and specific-instanceof both carry through, and the // JSON.parse throws SyntaxError: name and specific-instanceof both carry through, and the
// message keeps the engine's position detail. // message keeps the engine's position detail.
expect(await value(` expect(
await value(`
try { JSON.parse("{oops") } catch (e) { try { JSON.parse("{oops") } catch (e) {
return [e.name, e instanceof SyntaxError, e instanceof Error, e instanceof TypeError, e.message.includes("JSON")] return [e.name, e instanceof SyntaxError, e instanceof Error, e instanceof TypeError, e.message.includes("JSON")]
} }
`)).toEqual(["SyntaxError", true, true, false, true]) `),
expect(await value(`try { undeclared() } catch (e) { return [e.name, e instanceof ReferenceError] }`)) ).toEqual(["SyntaxError", true, true, false, true])
.toEqual(["ReferenceError", true]) expect(await value(`try { undeclared() } catch (e) { return [e.name, e instanceof ReferenceError] }`)).toEqual([
expect(await value(`try { const c = 1; c = 2 } catch (e) { return [e.name, e instanceof TypeError] }`)) "ReferenceError",
.toEqual(["TypeError", true]) true,
expect(await value(`try { "a".normalize("NOPE") } catch (e) { return [e.name, e instanceof RangeError] }`)) ])
.toEqual(["RangeError", true]) expect(await value(`try { const c = 1; c = 2 } catch (e) { return [e.name, e instanceof TypeError] }`)).toEqual([
expect(await value(`try { "a".match("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)) "TypeError",
.toEqual(["SyntaxError", true]) true,
expect(await value(`try { new RegExp("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)) ])
.toEqual(["SyntaxError", true]) expect(await value(`try { "a".normalize("NOPE") } catch (e) { return [e.name, e instanceof RangeError] }`)).toEqual(
["RangeError", true],
)
expect(await value(`try { "a".match("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)).toEqual([
"SyntaxError",
true,
])
expect(await value(`try { new RegExp("(") } catch (e) { return [e.name, e instanceof SyntaxError] }`)).toEqual([
"SyntaxError",
true,
])
}) })
test("diagnostics without a specific real-JS analogue are named plain Error", async () => { test("diagnostics without a specific real-JS analogue are named plain Error", async () => {
expect(await value(`try { JSON.parse(5) } catch (e) { return [e.name, e instanceof Error] }`)) expect(await value(`try { JSON.parse(5) } catch (e) { return [e.name, e instanceof Error] }`)).toEqual([
.toEqual(["Error", true]) "Error",
true,
])
}) })
test("Promise.allSettled rejection reasons are Error values", async () => { test("Promise.allSettled rejection reasons are Error values", async () => {
expect(await value(` expect(
await value(`
const settled = await Promise.allSettled([Promise.reject(new Error("b"))]) const settled = await Promise.allSettled([Promise.reject(new Error("b"))])
return [settled[0].reason instanceof Error, settled[0].reason.message] return [settled[0].reason instanceof Error, settled[0].reason.message]
`)).toEqual([true, "b"]) `),
).toEqual([true, "b"])
}) })
test("non-error thrown values are not instanceof Error", async () => { test("non-error thrown values are not instanceof Error", async () => {
@ -203,7 +237,11 @@ describe("Error values and instanceof", () => {
}) })
test("plain data is never instanceof Error", async () => { test("plain data is never instanceof Error", async () => {
expect(await value(`return [({}) instanceof Error, "s" instanceof Error, null instanceof Error]`)).toEqual([false, false, false]) expect(await value(`return [({}) instanceof Error, "s" instanceof Error, null instanceof Error]`)).toEqual([
false,
false,
false,
])
}) })
test("error values still serialize as plain { name, message } data", async () => { test("error values still serialize as plain { name, message } data", async () => {
@ -227,17 +265,29 @@ describe("Error values and instanceof", () => {
describe("array methods: splice, fill, copyWithin, keys/values/entries", () => { describe("array methods: splice, fill, copyWithin, keys/values/entries", () => {
test("splice removes in place and returns the removed elements", async () => { test("splice removes in place and returns the removed elements", async () => {
expect(await value(`const a = [1,2,3,4]; const removed = a.splice(1, 2); return { removed, a }`)).toEqual({ removed: [2, 3], a: [1, 4] }) expect(await value(`const a = [1,2,3,4]; const removed = a.splice(1, 2); return { removed, a }`)).toEqual({
removed: [2, 3],
a: [1, 4],
})
}) })
test("splice inserts new elements at the cut", async () => { test("splice inserts new elements at the cut", async () => {
expect(await value(`const a = ["a","d"]; a.splice(1, 0, "b", "c"); return a`)).toEqual(["a", "b", "c", "d"]) expect(await value(`const a = ["a","d"]; a.splice(1, 0, "b", "c"); return a`)).toEqual(["a", "b", "c", "d"])
expect(await value(`const a = [1,2,3]; const removed = a.splice(1, 1, "x"); return { removed, a }`)).toEqual({ removed: [2], a: [1, "x", 3] }) expect(await value(`const a = [1,2,3]; const removed = a.splice(1, 1, "x"); return { removed, a }`)).toEqual({
removed: [2],
a: [1, "x", 3],
})
}) })
test("splice with one argument removes to the end; negative start counts back", async () => { test("splice with one argument removes to the end; negative start counts back", async () => {
expect(await value(`const a = [1,2,3]; const removed = a.splice(1); return { removed, a }`)).toEqual({ removed: [2, 3], a: [1] }) expect(await value(`const a = [1,2,3]; const removed = a.splice(1); return { removed, a }`)).toEqual({
expect(await value(`const a = [1,2,3]; const removed = a.splice(-1); return { removed, a }`)).toEqual({ removed: [3], a: [1, 2] }) removed: [2, 3],
a: [1],
})
expect(await value(`const a = [1,2,3]; const removed = a.splice(-1); return { removed, a }`)).toEqual({
removed: [3],
a: [1, 2],
})
}) })
test("splice rejects inserting a container into itself", async () => { test("splice rejects inserting a container into itself", async () => {
@ -258,11 +308,13 @@ describe("array methods: splice, fill, copyWithin, keys/values/entries", () => {
test("keys/values/entries return arrays usable with for...of and spread", async () => { test("keys/values/entries return arrays usable with for...of and spread", async () => {
expect(await value(`return [...["x","y","z"].keys()]`)).toEqual([0, 1, 2]) expect(await value(`return [...["x","y","z"].keys()]`)).toEqual([0, 1, 2])
expect(await value(`return ["x","y"].values()`)).toEqual(["x", "y"]) expect(await value(`return ["x","y"].values()`)).toEqual(["x", "y"])
expect(await value(` expect(
await value(`
const out = [] const out = []
for (const [index, item] of ["a","b"].entries()) out.push(index + ":" + item) for (const [index, item] of ["a","b"].entries()) out.push(index + ":" + item)
return out return out
`)).toEqual(["0:a", "1:b"]) `),
).toEqual(["0:a", "1:b"])
expect(await value(`return [...[7].entries()]`)).toEqual([[0, 7]]) expect(await value(`return [...[7].entries()]`)).toEqual([[0, 7]])
}) })
}) })
@ -300,40 +352,33 @@ describe("compound assignment matches its binary operator", () => {
} }
test("sandbox Date += concatenates its string form, like d = d + 1", async () => { test("sandbox Date += concatenates its string form, like d = d + 1", async () => {
const result = await pair( const result = await pair(`let d = new Date(1000); d += 1; return d`, `let d = new Date(1000); d = d + 1; return d`)
`let d = new Date(1000); d += 1; return d`,
`let d = new Date(1000); d = d + 1; return d`,
)
expect(result).toBe("1970-01-01T00:00:01.000Z1") expect(result).toBe("1970-01-01T00:00:01.000Z1")
}) })
test("sandbox Date numeric compound ops use its time value", async () => { test("sandbox Date numeric compound ops use its time value", async () => {
expect(await pair( expect(
`let d = new Date(1000); d -= 400; return d`, await pair(`let d = new Date(1000); d -= 400; return d`, `let d = new Date(1000); d = d - 400; return d`),
`let d = new Date(1000); d = d - 400; return d`, ).toBe(600)
)).toBe(600) expect(await pair(`let d = new Date(1000); d /= 4; return d`, `let d = new Date(1000); d = d / 4; return d`)).toBe(
expect(await pair( 250,
`let d = new Date(1000); d /= 4; return d`, )
`let d = new Date(1000); d = d / 4; return d`,
)).toBe(250)
}) })
test("string += object/array matches x = x + obj", async () => { test("string += object/array matches x = x + obj", async () => {
expect(await pair( expect(await pair(`let x = "a"; x += { b: 1 }; return x`, `let x = "a"; x = x + { b: 1 }; return x`)).toBe(
`let x = "a"; x += { b: 1 }; return x`, "a[object Object]",
`let x = "a"; x = x + { b: 1 }; return x`, )
)).toBe("a[object Object]") expect(await pair(`let x = "a"; x += [1, 2]; return x`, `let x = "a"; x = x + [1, 2]; return x`)).toBe("a1,2")
expect(await pair(
`let x = "a"; x += [1, 2]; return x`,
`let x = "a"; x = x + [1, 2]; return x`,
)).toBe("a1,2")
}) })
test("compound assignment through a member target coerces the same way", async () => { test("compound assignment through a member target coerces the same way", async () => {
expect(await pair( expect(
`const o = { s: "t" }; o.s += new Date(0); return o.s`, await pair(
`const o = { s: "t" }; o.s = o.s + new Date(0); return o.s`, `const o = { s: "t" }; o.s += new Date(0); return o.s`,
)).toBe("t1970-01-01T00:00:00.000Z") `const o = { s: "t" }; o.s = o.s + new Date(0); return o.s`,
),
).toBe("t1970-01-01T00:00:00.000Z")
}) })
test("numeric and string compound operators sweep identically to their expansions", async () => { test("numeric and string compound operators sweep identically to their expansions", async () => {

View file

@ -23,7 +23,7 @@ const sleepyTool = (trace: Trace) =>
input: Schema.Struct({ id: Schema.Number, ms: Schema.optionalKey(Schema.Number) }), input: Schema.Struct({ id: Schema.Number, ms: Schema.optionalKey(Schema.Number) }),
output: Schema.Number, output: Schema.Number,
run: ({ id, ms }) => run: ({ id, ms }) =>
Effect.gen(function*() { Effect.gen(function* () {
trace.starts.push(id) trace.starts.push(id)
trace.active += 1 trace.active += 1
trace.maxActive = Math.max(trace.maxActive, trace.active) trace.maxActive = Math.max(trace.maxActive, trace.active)
@ -31,10 +31,14 @@ const sleepyTool = (trace: Trace) =>
trace.active -= 1 trace.active -= 1
trace.completed += 1 trace.completed += 1
return id return id
}).pipe(Effect.onInterrupt(() => Effect.sync(() => { }).pipe(
trace.active -= 1 Effect.onInterrupt(() =>
trace.interrupted += 1 Effect.sync(() => {
}))), trace.active -= 1
trace.interrupted += 1
}),
),
),
}) })
const failingTool = Tool.make({ const failingTool = Tool.make({
@ -46,11 +50,13 @@ const failingTool = Tool.make({
const run = (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}): Promise<ExecuteResult> => { const run = (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}): Promise<ExecuteResult> => {
const trace = options.trace ?? makeTrace() const trace = options.trace ?? makeTrace()
return Effect.runPromise(CodeMode.execute({ return Effect.runPromise(
tools: { host: { sleepy: sleepyTool(trace), fail: failingTool } }, CodeMode.execute({
code, tools: { host: { sleepy: sleepyTool(trace), fail: failingTool } },
...(options.limits ? { limits: options.limits } : {}), code,
})) ...(options.limits ? { limits: options.limits } : {}),
}),
)
} }
const value = async (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}) => { const value = async (code: string, options: { trace?: Trace; limits?: ExecutionLimits } = {}) => {
@ -121,7 +127,8 @@ describe("first-class promise values", () => {
}) })
test("an awaited failure is catchable exactly like a synchronous throw", async () => { test("an awaited failure is catchable exactly like a synchronous throw", async () => {
expect(await value(` expect(
await value(`
const p = tools.host.fail({}) const p = tools.host.fail({})
try { try {
await p await p
@ -129,7 +136,8 @@ describe("first-class promise values", () => {
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a fire-and-forget call completes before the execution ends", async () => { test("a fire-and-forget call completes before the execution ends", async () => {
@ -186,20 +194,24 @@ describe("promises at data boundaries", () => {
describe("Promise.all over arbitrary arrays", () => { describe("Promise.all over arbitrary arrays", () => {
test("mixes promises and plain values, preserving order", async () => { test("mixes promises and plain values, preserving order", async () => {
expect(await value(` expect(
await value(`
return await Promise.all([tools.host.sleepy({ id: 1 }), "plain", tools.host.sleepy({ id: 2 }), 42]) return await Promise.all([tools.host.sleepy({ id: 1 }), "plain", tools.host.sleepy({ id: 2 }), 42])
`)).toEqual([1, "plain", 2, 42]) `),
).toEqual([1, "plain", 2, 42])
}) })
test("accepts arrays built beforehand, passed as identifiers, and spread elements", async () => { test("accepts arrays built beforehand, passed as identifiers, and spread elements", async () => {
expect(await value(` expect(
await value(`
const calls = [] const calls = []
calls.push(tools.host.sleepy({ id: 1 })) calls.push(tools.host.sleepy({ id: 1 }))
calls.push(7) calls.push(7)
const more = [tools.host.sleepy({ id: 2 })] const more = [tools.host.sleepy({ id: 2 })]
const batch = [...calls, ...more, "x"] const batch = [...calls, ...more, "x"]
return await Promise.all(batch) return await Promise.all(batch)
`)).toEqual([1, 7, 2, "x"]) `),
).toEqual([1, 7, 2, "x"])
}) })
test("runs items.map tool calls in parallel", async () => { test("runs items.map tool calls in parallel", async () => {
@ -238,14 +250,16 @@ describe("Promise.all over arbitrary arrays", () => {
}) })
test("rejects with the first failure, catchable in-program", async () => { test("rejects with the first failure, catchable in-program", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.all([tools.host.sleepy({ id: 1 }), tools.host.fail({})]) await Promise.all([tools.host.sleepy({ id: 1 }), tools.host.fail({})])
return "no" return "no"
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a non-collection argument is a clear error", async () => { test("a non-collection argument is a clear error", async () => {
@ -264,14 +278,16 @@ describe("Promise.all over arbitrary arrays", () => {
describe("Promise.allSettled", () => { describe("Promise.allSettled", () => {
test("reports fulfilled and rejected outcomes with catch-normalized reasons", async () => { test("reports fulfilled and rejected outcomes with catch-normalized reasons", async () => {
expect(await value(` expect(
await value(`
return await Promise.allSettled([ return await Promise.allSettled([
tools.host.sleepy({ id: 5 }), tools.host.sleepy({ id: 5 }),
tools.host.fail({}), tools.host.fail({}),
"plain", "plain",
Promise.reject(new Error("boom")), Promise.reject(new Error("boom")),
]) ])
`)).toEqual([ `),
).toEqual([
{ status: "fulfilled", value: 5 }, { status: "fulfilled", value: 5 },
{ status: "rejected", reason: { name: "Error", message: "Lookup refused" } }, { status: "rejected", reason: { name: "Error", message: "Lookup refused" } },
{ status: "fulfilled", value: "plain" }, { status: "fulfilled", value: "plain" },
@ -306,7 +322,8 @@ describe("Promise.race", () => {
}) })
test("awaiting an interrupted loser afterwards is a catchable program failure", async () => { test("awaiting an interrupted loser afterwards is a catchable program failure", async () => {
expect(await value(` expect(
await value(`
const fast = tools.host.sleepy({ id: 1, ms: 10 }) const fast = tools.host.sleepy({ id: 1, ms: 10 })
const slow = tools.host.sleepy({ id: 2, ms: 5000 }) const slow = tools.host.sleepy({ id: 2, ms: 5000 })
const winner = await Promise.race([fast, slow]) const winner = await Promise.race([fast, slow])
@ -316,23 +333,31 @@ describe("Promise.race", () => {
} catch (e) { } catch (e) {
return { winner, caught: e.message } return { winner, caught: e.message }
} }
`)).toEqual({ winner: 1, caught: "This tool call was interrupted because another value settled a Promise.race first." }) `),
).toEqual({
winner: 1,
caught: "This tool call was interrupted because another value settled a Promise.race first.",
})
}) })
test("a rejection can win the race", async () => { test("a rejection can win the race", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.race([tools.host.fail({}), tools.host.sleepy({ id: 1, ms: 5000 })]) await Promise.race([tools.host.fail({}), tools.host.sleepy({ id: 1, ms: 5000 })])
return "no" return "no"
} catch (e) { } catch (e) {
return e.message return e.message
} }
`)).toBe("Lookup refused") `),
).toBe("Lookup refused")
}) })
test("a plain value wins over pending promises", async () => { test("a plain value wins over pending promises", async () => {
const trace = makeTrace() const trace = makeTrace()
expect(await value(`return await Promise.race([tools.host.sleepy({ id: 1, ms: 5000 }), "immediate"])`, { trace })).toBe("immediate") expect(
await value(`return await Promise.race([tools.host.sleepy({ id: 1, ms: 5000 }), "immediate"])`, { trace }),
).toBe("immediate")
expect(trace.interrupted).toBe(1) expect(trace.interrupted).toBe(1)
}) })
@ -350,14 +375,16 @@ describe("Promise.resolve / Promise.reject", () => {
}) })
test("reject produces a promise whose await throws the reason", async () => { test("reject produces a promise whose await throws the reason", async () => {
expect(await value(` expect(
await value(`
try { try {
await Promise.reject("nope") await Promise.reject("nope")
return "no" return "no"
} catch (e) { } catch (e) {
return e return e
} }
`)).toBe("nope") `),
).toBe("nope")
}) })
}) })

View file

@ -83,15 +83,9 @@ describe("pretty signature rendering", () => {
true, true,
) )
expect(pretty).toBe( expect(pretty).toBe(
[ ["{", " /** Search filter */", " filter?: {", " /** Issue state */", " state?: string", " }", "}"].join(
"{", "\n",
" /** Search filter */", ),
" filter?: {",
" /** Issue state */",
" state?: string",
" }",
"}",
].join("\n"),
) )
}) })
@ -119,7 +113,14 @@ describe("pretty signature rendering", () => {
expect(pretty).toContain(" /** @deprecated */\n legacy?: string") expect(pretty).toContain(" /** @deprecated */\n legacy?: string")
expect(pretty).toContain(" /** @format uri */\n homepage?: string") expect(pretty).toContain(" /** @format uri */\n homepage?: string")
expect(pretty).toContain( expect(pretty).toContain(
[" /**", ' * @default ["a","b"]', " * @minItems 2", " * @maxItems 5", " */", " tags?: Array<string>"].join("\n"), [
" /**",
' * @default ["a","b"]',
" * @minItems 2",
" * @maxItems 5",
" */",
" tags?: Array<string>",
].join("\n"),
) )
}) })
@ -212,7 +213,11 @@ describe("non-identifier property names render as quoted keys", () => {
const tool = Tool.make({ const tool = Tool.make({
description: "Adapter tool with awkward field names", description: "Adapter tool with awkward field names",
input: rawSchema, input: rawSchema,
output: { type: "object", properties: { "content-type": { type: "string" } }, required: ["content-type"] } as const, output: {
type: "object",
properties: { "content-type": { type: "string" } },
required: ["content-type"],
} as const,
run: () => Effect.succeed({ "content-type": "text/plain" }), run: () => Effect.succeed({ "content-type": "text/plain" }),
}) })
expect(inputTypeScript(tool)).toContain('"foo-bar"?: string') expect(inputTypeScript(tool)).toContain('"foo-bar"?: string')
@ -269,9 +274,9 @@ describe("pretty signatures in search results", () => {
const runtime = CodeMode.make({ tools: { github: { list_issues: listIssues }, orders: { lookup: lookupOrder } } }) const runtime = CodeMode.make({ tools: { github: { list_issues: listIssues }, orders: { lookup: lookupOrder } } })
const search = async (query: string) => { const search = async (query: string) => {
const result = await Effect.runPromise(runtime.execute( const result = await Effect.runPromise(
`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`, runtime.execute(`return await tools.$codemode.search({ query: ${JSON.stringify(query)} })`),
)) )
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
if (!result.ok) throw new Error("search failed") if (!result.ok) throw new Error("search failed")
return result.value as { items: Array<{ path: string; signature: string }>; total: number } return result.value as { items: Array<{ path: string; signature: string }>; total: number }

View file

@ -40,7 +40,11 @@ describe("Date", () => {
}) })
test("UTC getters read calendar components", async () => { test("UTC getters read calendar components", async () => {
expect(await value(`const d = new Date("2024-03-05T06:07:08.009Z"); return [d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate(), d.getUTCHours(), d.getUTCMinutes(), d.getUTCSeconds(), d.getUTCMilliseconds()]`)).toEqual([2024, 2, 5, 6, 7, 8, 9]) expect(
await value(
`const d = new Date("2024-03-05T06:07:08.009Z"); return [d.getUTCFullYear(), d.getUTCMonth(), d.getUTCDate(), d.getUTCHours(), d.getUTCMinutes(), d.getUTCSeconds(), d.getUTCMilliseconds()]`,
),
).toEqual([2024, 2, 5, 6, 7, 8, 9])
}) })
test("invalid dates yield NaN times, guardable in-sandbox", async () => { test("invalid dates yield NaN times, guardable in-sandbox", async () => {
@ -49,7 +53,9 @@ describe("Date", () => {
}) })
test("toISOString on an invalid date is a catchable error", async () => { test("toISOString on an invalid date is a catchable error", async () => {
expect(await value(`try { new Date("garbage").toISOString(); return "no" } catch { return "caught" }`)).toBe("caught") expect(await value(`try { new Date("garbage").toISOString(); return "no" } catch { return "caught" }`)).toBe(
"caught",
)
}) })
test("template interpolation renders the ISO form", async () => { test("template interpolation renders the ISO form", async () => {
@ -72,14 +78,18 @@ describe("Date", () => {
}) })
test("sorting dates with a numeric comparator", async () => { test("sorting dates with a numeric comparator", async () => {
expect(await value(` expect(
await value(`
const dates = [new Date(3000), new Date(1000), new Date(2000)] const dates = [new Date(3000), new Date(1000), new Date(2000)]
return dates.sort((a, b) => a - b).map((d) => d.getTime()) return dates.sort((a, b) => a - b).map((d) => d.getTime())
`)).toEqual([1000, 2000, 3000]) `),
).toEqual([1000, 2000, 3000])
}) })
test("new Date(year, month, day) accepts component form", async () => { test("new Date(year, month, day) accepts component form", async () => {
expect(await value(`const d = new Date(2024, 0, 2); return [d.getFullYear(), d.getMonth(), d.getDate()]`)).toEqual([2024, 0, 2]) expect(await value(`const d = new Date(2024, 0, 2); return [d.getFullYear(), d.getMonth(), d.getDate()]`)).toEqual([
2024, 0, 2,
])
}) })
test("typeof and unknown properties are forgiving", async () => { test("typeof and unknown properties are forgiving", async () => {
@ -95,25 +105,31 @@ describe("RegExp", () => {
}) })
test("exec exposes captures and index", async () => { test("exec exposes captures and index", async () => {
expect(await value(`const m = /a(b+)/.exec("xxabbc"); return { full: m[0], group: m[1], index: m.index }`)).toEqual({ expect(await value(`const m = /a(b+)/.exec("xxabbc"); return { full: m[0], group: m[1], index: m.index }`)).toEqual(
full: "abb", {
group: "bb", full: "abb",
index: 2, group: "bb",
}) index: 2,
},
)
expect(await value(`return /a/.exec("zzz")`)).toBeNull() expect(await value(`return /a/.exec("zzz")`)).toBeNull()
}) })
test("named groups read through", async () => { test("named groups read through", async () => {
expect(await value(`const m = /(?<word>[a-z]+)-(?<num>\\d+)/.exec("id ab-42"); return m.groups.word + m.groups.num`)).toBe("ab42") expect(
await value(`const m = /(?<word>[a-z]+)-(?<num>\\d+)/.exec("id ab-42"); return m.groups.word + m.groups.num`),
).toBe("ab42")
}) })
test("global exec advances lastIndex across calls", async () => { test("global exec advances lastIndex across calls", async () => {
expect(await value(` expect(
await value(`
const r = /\\d+/g const r = /\\d+/g
const first = r.exec("a1b22c") const first = r.exec("a1b22c")
const second = r.exec("a1b22c") const second = r.exec("a1b22c")
return [first[0], second[0]] return [first[0], second[0]]
`)).toEqual(["1", "22"]) `),
).toEqual(["1", "22"])
}) })
test("string match: non-global carries index, global lists all matches", async () => { test("string match: non-global carries index, global lists all matches", async () => {
@ -194,34 +210,51 @@ describe("RegExp", () => {
describe("Map", () => { describe("Map", () => {
test("get/set/has/size with chaining", async () => { test("get/set/has/size with chaining", async () => {
expect(await value(` expect(
await value(`
const m = new Map() const m = new Map()
m.set("a", 1).set("b", 2) m.set("a", 1).set("b", 2)
return { a: m.get("a"), b: m.get("b"), has: m.has("a"), miss: m.get("zz") === undefined, size: m.size } return { a: m.get("a"), b: m.get("b"), has: m.has("a"), miss: m.get("zz") === undefined, size: m.size }
`)).toEqual({ a: 1, b: 2, has: true, miss: true, size: 5 - 3 }) `),
).toEqual({ a: 1, b: 2, has: true, miss: true, size: 5 - 3 })
}) })
test("object keys use identity", async () => { test("object keys use identity", async () => {
expect(await value(` expect(
await value(`
const key = { id: 1 } const key = { id: 1 }
const m = new Map() const m = new Map()
m.set(key, "hit") m.set(key, "hit")
return [m.get(key), m.get({ id: 1 }) === undefined] return [m.get(key), m.get({ id: 1 }) === undefined]
`)).toEqual(["hit", true]) `),
).toEqual(["hit", true])
}) })
test("construction from entry pairs and another Map", async () => { test("construction from entry pairs and another Map", async () => {
expect(await value(`const m = new Map([["a", 1], ["b", 2]]); return m.get("b")`)).toBe(2) expect(await value(`const m = new Map([["a", 1], ["b", 2]]); return m.get("b")`)).toBe(2)
expect(await value(`const m = new Map([["a", 1]]); const n = new Map(m); n.set("b", 2); return [n.get("a"), n.get("b"), m.has("b")]`)).toEqual([1, 2, false]) expect(
await value(
`const m = new Map([["a", 1]]); const n = new Map(m); n.set("b", 2); return [n.get("a"), n.get("b"), m.has("b")]`,
),
).toEqual([1, 2, false])
expect((await error(`return new Map("nope")`)).message).toMatch(/\[key, value\] pairs/) expect((await error(`return new Map("nope")`)).message).toMatch(/\[key, value\] pairs/)
expect((await error(`return new Map(["flat"])`)).message).toMatch(/\[key, value\] pairs/) expect((await error(`return new Map(["flat"])`)).message).toMatch(/\[key, value\] pairs/)
}) })
test("keys/values/entries return arrays", async () => { test("keys/values/entries return arrays", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
return { keys: m.keys(), values: m.values(), entries: m.entries() } return { keys: m.keys(), values: m.values(), entries: m.entries() }
`)).toEqual({ keys: ["a", "b"], values: [1, 2], entries: [["a", 1], ["b", 2]] }) `),
).toEqual({
keys: ["a", "b"],
values: [1, 2],
entries: [
["a", 1],
["b", 2],
],
})
}) })
test("Object.fromEntries(map) and Array.from(map)", async () => { test("Object.fromEntries(map) and Array.from(map)", async () => {
@ -230,13 +263,15 @@ describe("Map", () => {
}) })
test("for...of iterates [key, value] pairs with destructuring", async () => { test("for...of iterates [key, value] pairs with destructuring", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
let total = 0 let total = 0
let names = "" let names = ""
for (const [key, count] of m) { names += key; total += count } for (const [key, count] of m) { names += key; total += count }
return names + total return names + total
`)).toBe("ab3") `),
).toBe("ab3")
}) })
test("spread produces entry pairs", async () => { test("spread produces entry pairs", async () => {
@ -244,32 +279,38 @@ describe("Map", () => {
}) })
test("forEach passes (value, key)", async () => { test("forEach passes (value, key)", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
const seen = [] const seen = []
m.forEach((count, key) => seen.push(key + count)) m.forEach((count, key) => seen.push(key + count))
return seen return seen
`)).toEqual(["a1", "b2"]) `),
).toEqual(["a1", "b2"])
}) })
test("delete and clear", async () => { test("delete and clear", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["a", 1], ["b", 2]]) const m = new Map([["a", 1], ["b", 2]])
const removed = m.delete("a") const removed = m.delete("a")
const missed = m.delete("zz") const missed = m.delete("zz")
const sizeAfterDelete = m.size const sizeAfterDelete = m.size
m.clear() m.clear()
return [removed, missed, sizeAfterDelete, m.size] return [removed, missed, sizeAfterDelete, m.size]
`)).toEqual([true, false, 1, 0]) `),
).toEqual([true, false, 1, 0])
}) })
test("counting idiom: grouped tallies", async () => { test("counting idiom: grouped tallies", async () => {
expect(await value(` expect(
await value(`
const words = ["a", "b", "a", "c", "a"] const words = ["a", "b", "a", "c", "a"]
const counts = new Map() const counts = new Map()
for (const word of words) counts.set(word, (counts.get(word) ?? 0) + 1) for (const word of words) counts.set(word, (counts.get(word) ?? 0) + 1)
return Object.fromEntries(counts) return Object.fromEntries(counts)
`)).toEqual({ a: 3, b: 1, c: 1 }) `),
).toEqual({ a: 3, b: 1, c: 1 })
}) })
test("maps serialize to {} at the boundary, like JSON", async () => { test("maps serialize to {} at the boundary, like JSON", async () => {
@ -286,12 +327,14 @@ describe("Map", () => {
describe("Set", () => { describe("Set", () => {
test("add/has/delete/size with chaining", async () => { test("add/has/delete/size with chaining", async () => {
expect(await value(` expect(
await value(`
const s = new Set() const s = new Set()
s.add(1).add(2).add(1) s.add(1).add(2).add(1)
const removed = s.delete(2) const removed = s.delete(2)
return [s.size, s.has(1), s.has(2), removed] return [s.size, s.has(1), s.has(2), removed]
`)).toEqual([1, true, false, true]) `),
).toEqual([1, true, false, true])
}) })
test("dedupe idiom: [...new Set(items)]", async () => { test("dedupe idiom: [...new Set(items)]", async () => {
@ -308,11 +351,13 @@ describe("Set", () => {
}) })
test("for...of iterates values", async () => { test("for...of iterates values", async () => {
expect(await value(` expect(
await value(`
let total = 0 let total = 0
for (const n of new Set([1, 2, 3])) total += n for (const n of new Set([1, 2, 3])) total += n
return total return total
`)).toBe(6) `),
).toBe(6)
}) })
test("sets serialize to {} at the boundary, like JSON", async () => { test("sets serialize to {} at the boundary, like JSON", async () => {
@ -338,21 +383,32 @@ describe("stdlib integration", () => {
}) })
test("dates inside Map values survive in-sandbox reads", async () => { test("dates inside Map values survive in-sandbox reads", async () => {
expect(await value(` expect(
await value(`
const m = new Map([["start", new Date(1000)]]) const m = new Map([["start", new Date(1000)]])
return m.get("start").getTime() return m.get("start").getTime()
`)).toBe(1000) `),
).toBe(1000)
}) })
test("instanceof recognizes the stdlib value types", async () => { test("instanceof recognizes the stdlib value types", async () => {
expect(await value(`return [new Date(0) instanceof Date, /a/ instanceof RegExp, new Map() instanceof Map, new Set() instanceof Set]`)).toEqual([true, true, true, true]) expect(
expect(await value(`return [[1] instanceof Array, [1] instanceof Object, ({}) instanceof Object, 5 instanceof Object]`)).toEqual([true, true, true, false]) await value(
`return [new Date(0) instanceof Date, /a/ instanceof RegExp, new Map() instanceof Map, new Set() instanceof Set]`,
),
).toEqual([true, true, true, true])
expect(
await value(`return [[1] instanceof Array, [1] instanceof Object, ({}) instanceof Object, 5 instanceof Object]`),
).toEqual([true, true, true, false])
expect(await value(`return [new Map() instanceof Set, "s" instanceof Date]`)).toEqual([false, false]) expect(await value(`return [new Map() instanceof Set, "s" instanceof Date]`)).toEqual([false, false])
expect(await value(`const p = Promise.resolve(1); const isPromise = p instanceof Promise; await p; return isPromise`)).toBe(true) expect(
await value(`const p = Promise.resolve(1); const isPromise = p instanceof Promise; await p; return isPromise`),
).toBe(true)
}) })
test("realistic pipeline: parse, extract with regex, dedupe, count by day", async () => { test("realistic pipeline: parse, extract with regex, dedupe, count by day", async () => {
expect(await value(` expect(
await value(`
const raw = '[{"at":"2024-01-01T05:00:00Z","tag":"a b"},{"at":"2024-01-01T09:00:00Z","tag":"b c"},{"at":"2024-01-02T01:00:00Z","tag":"a"}]' const raw = '[{"at":"2024-01-01T05:00:00Z","tag":"a b"},{"at":"2024-01-01T09:00:00Z","tag":"b c"},{"at":"2024-01-02T01:00:00Z","tag":"a"}]'
const rows = JSON.parse(raw) const rows = JSON.parse(raw)
const tags = new Set() const tags = new Set()
@ -363,27 +419,34 @@ describe("stdlib integration", () => {
byDay.set(day, (byDay.get(day) ?? 0) + 1) byDay.set(day, (byDay.get(day) ?? 0) + 1)
} }
return { tags: [...tags].sort((a, b) => (a < b ? -1 : 1)), byDay: Object.fromEntries(byDay) } return { tags: [...tags].sort((a, b) => (a < b ? -1 : 1)), byDay: Object.fromEntries(byDay) }
`)).toEqual({ tags: ["a", "b", "c"], byDay: { "2024-01-01": 2, "2024-01-02": 1 } }) `),
).toEqual({ tags: ["a", "b", "c"], byDay: { "2024-01-01": 2, "2024-01-02": 1 } })
}) })
}) })
describe("sandbox values at intra-sandbox checkpoints", () => { describe("sandbox values at intra-sandbox checkpoints", () => {
test("Object.values/entries keep Dates usable", async () => { test("Object.values/entries keep Dates usable", async () => {
expect(await value(`return Object.values({ d: new Date(0) })[0].getTime()`)).toBe(0) expect(await value(`return Object.values({ d: new Date(0) })[0].getTime()`)).toBe(0)
expect(await value(`const [key, d] = Object.entries({ d: new Date(0) })[0]; return key + ":" + d.getTime()`)).toBe("d:0") expect(await value(`const [key, d] = Object.entries({ d: new Date(0) })[0]; return key + ":" + d.getTime()`)).toBe(
"d:0",
)
}) })
test("Object.assign keeps Maps usable", async () => { test("Object.assign keeps Maps usable", async () => {
expect(await value(`const merged = Object.assign({}, { m: new Map([["a", 1]]) }); return merged.m.get("a")`)).toBe(1) expect(await value(`const merged = Object.assign({}, { m: new Map([["a", 1]]) }); return merged.m.get("a")`)).toBe(
1,
)
}) })
test("object and array spread keep sandbox values usable", async () => { test("object and array spread keep sandbox values usable", async () => {
expect(await value(` expect(
await value(`
const src = { m: new Map([["a", 1]]) } const src = { m: new Map([["a", 1]]) }
const copy = { ...src } const copy = { ...src }
copy.m.set("b", 2) copy.m.set("b", 2)
return [copy.m.get("a"), src.m.get("b")] return [copy.m.get("a"), src.m.get("b")]
`)).toEqual([1, 2]) `),
).toEqual([1, 2])
expect(await value(`const list = [new Date(1000)]; const copy = [...list]; return copy[0].getTime()`)).toBe(1000) expect(await value(`const list = [new Date(1000)]; const copy = [...list]; return copy[0].getTime()`)).toBe(1000)
}) })
@ -404,7 +467,10 @@ describe("sandbox values at intra-sandbox checkpoints", () => {
}) })
test("the host boundary still serializes JSON forms: results, JSON.stringify, and tool arguments", async () => { test("the host boundary still serializes JSON forms: results, JSON.stringify, and tool arguments", async () => {
expect(await value(`return { d: new Date(0), m: new Map([["a", 1]]) }`)).toEqual({ d: "1970-01-01T00:00:00.000Z", m: {} }) expect(await value(`return { d: new Date(0), m: new Map([["a", 1]]) }`)).toEqual({
d: "1970-01-01T00:00:00.000Z",
m: {},
})
expect(await value(`return JSON.stringify({ d: new Date(0) })`)).toBe('{"d":"1970-01-01T00:00:00.000Z"}') expect(await value(`return JSON.stringify({ d: new Date(0) })`)).toBe('{"d":"1970-01-01T00:00:00.000Z"}')
const observed: Array<unknown> = [] const observed: Array<unknown> = []
@ -417,10 +483,12 @@ describe("sandbox values at intra-sandbox checkpoints", () => {
return "ok" return "ok"
}), }),
}) })
const result = await Effect.runPromise(CodeMode.execute({ const result = await Effect.runPromise(
tools: { host: { capture } }, CodeMode.execute({
code: `return await tools.host.capture({ when: new Date(0), tags: new Map([["a", 1]]) })`, tools: { host: { capture } },
})) code: `return await tools.host.capture({ when: new Date(0), tags: new Map([["a", 1]]) })`,
}),
)
expect(result.ok).toBe(true) expect(result.ok).toBe(true)
expect(observed).toStrictEqual([{ when: "1970-01-01T00:00:00.000Z", tags: {} }]) expect(observed).toStrictEqual([{ when: "1970-01-01T00:00:00.000Z", tags: {} }])
}) })

View file

@ -206,10 +206,7 @@ export const prepare = Effect.fn("LLMRequestPrep.prepare")(function* (input: Pre
}) })
function resolveTools(input: Pick<PrepareInput, "tools" | "agent" | "permission" | "user">) { function resolveTools(input: Pick<PrepareInput, "tools" | "agent" | "permission" | "user">) {
const visible = Permission.visibleTools( const visible = Permission.visibleTools(input.tools, Permission.merge(input.agent.permission, input.permission ?? []))
input.tools,
Permission.merge(input.agent.permission, input.permission ?? []),
)
return Record.filter(visible, (_, k) => input.user.tools?.[k] !== false) return Record.filter(visible, (_, k) => input.user.tools?.[k] !== false)
} }

View file

@ -104,7 +104,8 @@ export function groupByServer(
const byLongest = [...servers].sort((a, b) => b.length - a.length) const byLongest = [...servers].sort((a, b) => b.length - a.length)
const groups = new Map<string, CatalogEntry[]>() const groups = new Map<string, CatalogEntry[]>()
for (const key of Object.keys(mcpTools).sort((a, b) => a.localeCompare(b))) { for (const key of Object.keys(mcpTools).sort((a, b) => a.localeCompare(b))) {
const server = byLongest.find((name) => key.startsWith(name + "_")) ?? (key.includes("_") ? key.slice(0, key.indexOf("_")) : key) const server =
byLongest.find((name) => key.startsWith(name + "_")) ?? (key.includes("_") ? key.slice(0, key.indexOf("_")) : key)
const local = server && key.startsWith(server + "_") ? key.slice(server.length + 1) : key const local = server && key.startsWith(server + "_") ? key.slice(server.length + 1) : key
const def = mcpDefs[key] const def = mcpDefs[key]
const entry: CatalogEntry = { const entry: CatalogEntry = {
@ -129,7 +130,9 @@ export function buildCatalog(
mcpDefs: Record<string, MCPToolDef>, mcpDefs: Record<string, MCPToolDef>,
servers: readonly string[], servers: readonly string[],
): CatalogEntry[] { ): CatalogEntry[] {
return [...groupByServer(mcpTools, servers, mcpDefs).values()].flat().filter((entry) => entry.tool.execute !== undefined) return [...groupByServer(mcpTools, servers, mcpDefs).values()]
.flat()
.filter((entry) => entry.tool.execute !== undefined)
} }
/** /**
@ -334,7 +337,8 @@ export const CodeModeTool = Tool.define(
const collect = (attachment: Attachment) => void attachments.push(attachment) const collect = (attachment: Attachment) => void attachments.push(attachment)
// Stream the current call list to the UI. Sent on every status change so the // Stream the current call list to the UI. Sent on every status change so the
// tool part shows each child call appearing and resolving while the program runs. // tool part shows each child call appearing and resolving while the program runs.
const publish = () => ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } }) const publish = () =>
ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } })
// One CodeMode tool per MCP tool, running the same shared middle as legacy // One CodeMode tool per MCP tool, running the same shared middle as legacy
// per-tool registration (McpInvoke.invoke: plugin before hook → permission // per-tool registration (McpInvoke.invoke: plugin before hook → permission

View file

@ -276,7 +276,10 @@ const layer = Layer.effect(
// fresh per turn so it tracks live tool-list changes. Hard-denied tools (the shared // fresh per turn so it tracks live tool-list changes. Hard-denied tools (the shared
// Permission.visibleTools predicate over the agent's ruleset) never enter the // Permission.visibleTools predicate over the agent's ruleset) never enter the
// catalog, its inlined signatures, or the in-program search index. // catalog, its inlined signatures, or the in-program search index.
const describeCodeMode = Effect.fn("ToolRegistry.describeCodeMode")(function* (agent: Agent.Info, permission?: PermissionV1.Ruleset) { const describeCodeMode = Effect.fn("ToolRegistry.describeCodeMode")(function* (
agent: Agent.Info,
permission?: PermissionV1.Ruleset,
) {
const visible = Permission.visibleTools(yield* mcp.tools(), Permission.merge(agent.permission, permission ?? [])) const visible = Permission.visibleTools(yield* mcp.tools(), Permission.merge(agent.permission, permission ?? []))
const servers = Object.keys(yield* mcp.clients()).map(McpCatalog.sanitize) const servers = Object.keys(yield* mcp.clients()).map(McpCatalog.sanitize)
return catalogInstructions(visible, yield* mcp.defs(), servers) return catalogInstructions(visible, yield* mcp.defs(), servers)
@ -341,7 +344,6 @@ const layer = Layer.effect(
}), }),
) )
function isZodType(value: unknown): value is z.ZodType { function isZodType(value: unknown): value is z.ZodType {
return typeof value === "object" && value !== null && "_zod" in value return typeof value === "object" && value !== null && "_zod" in value
} }

View file

@ -96,8 +96,7 @@ function resolveTools(trigger?: Plugin.Interface["trigger"]) {
Layer.mergeAll( Layer.mergeAll(
Layer.mock(Permission.Service, { ask: () => Effect.void }), Layer.mock(Permission.Service, { ask: () => Effect.void }),
Layer.mock(Plugin.Service, { Layer.mock(Plugin.Service, {
trigger: trigger: trigger ?? (((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"]),
trigger ?? (((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"]),
}), }),
Layer.mock(Truncate.Service, { Layer.mock(Truncate.Service, {
output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), output: (text: string) => Effect.succeed({ content: text, truncated: false as const }),

View file

@ -21,8 +21,7 @@ import type { Tool as AITool } from "ai"
import { Effect, Layer } from "effect" import { Effect, Layer } from "effect"
// A 1x1 transparent PNG, base64-encoded, used to exercise image attachments. // A 1x1 transparent PNG, base64-encoded, used to exercise image attachments.
const PNG = const PNG = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
const SERVER = "fixtures" const SERVER = "fixtures"
@ -163,8 +162,8 @@ async function buildTool() {
// this real in-memory server listed — the same snapshot shape the live service returns. // this real in-memory server listed — the same snapshot shape the live service returns.
const layer = Layer.mergeAll( const layer = Layer.mergeAll(
Layer.mock(Plugin.Service, { Layer.mock(Plugin.Service, {
trigger: (((_name: unknown, _input: unknown, output: unknown) => trigger: ((_name: unknown, _input: unknown, output: unknown) =>
Effect.succeed(output)) as Plugin.Interface["trigger"]), Effect.succeed(output)) as Plugin.Interface["trigger"],
}), }),
Layer.mock(Truncate.Service, { Layer.mock(Truncate.Service, {
output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), output: (text: string) => Effect.succeed({ content: text, truncated: false as const }),
@ -196,9 +195,7 @@ describe("code mode integration (real MCP server)", () => {
test("the appended catalog inlines full signatures with real MCP schemas", () => { test("the appended catalog inlines full signatures with real MCP schemas", () => {
expect(description).toContain("Available tools (COMPLETE list") expect(description).toContain("Available tools (COMPLETE list")
expect(description).toContain("- fixtures (4 tools)") expect(description).toContain("- fixtures (4 tools)")
expect(description).toContain( expect(description).toContain("tools.fixtures.add(input: { a: number; b: number }): Promise<{ sum: number }>")
"tools.fixtures.add(input: { a: number; b: number }): Promise<{ sum: number }>",
)
expect(description).toContain("tools.fixtures.get_text(input: { name: string }): Promise<unknown>") expect(description).toContain("tools.fixtures.get_text(input: { name: string }): Promise<unknown>")
expect(description).toContain("// Add two numbers and return the structured sum") expect(description).toContain("// Add two numbers and return the structured sum")
// Small catalog: everything is inline, so no discovery tool is advertised. // Small catalog: everything is inline, so no discovery tool is advertised.

View file

@ -193,7 +193,9 @@ describe("code mode execute", () => {
// never cherry-picks a catalog tool or fabricates result fields. // never cherry-picks a catalog tool or fabricates result fields.
expect(description).toContain("## Workflow") expect(description).toContain("## Workflow")
expect(description).toContain("1. Pick a tool from the list under `## Available tools`") expect(description).toContain("1. Pick a tool from the list under `## Available tools`")
expect(description).toContain('`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string') expect(description).toContain(
'`const data = typeof res === "string" ? JSON.parse(res) : res` — most tools return JSON as a string',
)
expect(description).toContain("Return only the fields you need") expect(description).toContain("Return only the fields you need")
expect(description).not.toContain("total_count") expect(description).not.toContain("total_count")
}) })
@ -249,7 +251,9 @@ describe("code mode execute", () => {
expect(description).toContain("tools.$codemode.search(") expect(description).toContain("tools.$codemode.search(")
// PARTIAL catalogs put search first in the workflow and advertise namespace browsing. // PARTIAL catalogs put search first in the workflow and advertise namespace browsing.
expect(description).toContain("1. Find a tool (skip when it is already listed below)") expect(description).toContain("1. Find a tool (skip when it is already listed below)")
expect(description).toContain('- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.') expect(description).toContain(
'- Browse one namespace: `await tools.$codemode.search({ query: "", namespace: "<name>" })`.',
)
expect(description).not.toContain("total_count") expect(description).not.toContain("total_count")
// All op lines cost the same estimated tokens (chars/4 rounds away the 1- vs 3-digit // All op lines cost the same estimated tokens (chars/4 rounds away the 1- vs 3-digit
// name difference), so the path tiebreak decides: the lexicographically-first ops made // name difference), so the path tiebreak decides: the lexicographically-first ops made
@ -290,7 +294,10 @@ describe("code mode execute", () => {
linear_search: mcpTool("search", () => ""), linear_search: mcpTool("search", () => ""),
}) })
const output = await Effect.runPromise( const output = await Effect.runPromise(
tool.execute({ code: "const namespaces = Object.keys(tools); return { namespaces, count: namespaces.length }" }, ctx), tool.execute(
{ code: "const namespaces = Object.keys(tools); return { namespaces, count: namespaces.length }" },
ctx,
),
) )
expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear"], count: 2 }) expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear"], count: 2 })
}) })
@ -565,9 +572,7 @@ describe("code mode execute", () => {
const tool = await build({ const tool = await build({
shot_take: mcpTool("take", () => ({ content: [{ type: "image", data: "PNGDATA", mimeType: "image/png" }] })), shot_take: mcpTool("take", () => ({ content: [{ type: "image", data: "PNGDATA", mimeType: "image/png" }] })),
}) })
const out = await Effect.runPromise( const out = await Effect.runPromise(tool.execute({ code: "await tools.shot.take({}); return 'captured'" }, ctx))
tool.execute({ code: "await tools.shot.take({}); return 'captured'" }, ctx),
)
expect(out.output).toBe("captured") expect(out.output).toBe("captured")
expect(out.attachments).toHaveLength(1) expect(out.attachments).toHaveLength(1)
}) })
@ -690,9 +695,7 @@ describe("code mode permission visibility", () => {
expect(called).toEqual([]) expect(called).toEqual([])
// The rest of the namespace still works. // The rest of the namespace still works.
const allowed = await Effect.runPromise( const allowed = await Effect.runPromise(tool.execute({ code: "return await tools.github.list_issues({})" }, ctx))
tool.execute({ code: "return await tools.github.list_issues({})" }, ctx),
)
expect(allowed.metadata.error).toBeUndefined() expect(allowed.metadata.error).toBeUndefined()
expect(allowed.output).toBe("ok") expect(allowed.output).toBe("ok")
}) })
@ -706,9 +709,7 @@ describe("code mode permission visibility", () => {
["github"], ["github"],
[askRule("github_list_issues")], [askRule("github_list_issues")],
) )
const out = await Effect.runPromise( const out = await Effect.runPromise(tool.execute({ code: "return await tools.github.list_issues({})" }, askCtx))
tool.execute({ code: "return await tools.github.list_issues({})" }, askCtx),
)
expect(out.output).toBe("ok") expect(out.output).toBe("ok")
expect(asked).toEqual(["github_list_issues"]) expect(asked).toEqual(["github_list_issues"])
}) })
@ -733,16 +734,21 @@ describe("toSandboxResult", () => {
test("prefers structuredContent over text", () => { test("prefers structuredContent over text", () => {
const { collect } = collector() const { collect } = collector()
expect(toSandboxResult({ structuredContent: { x: 1 }, content: [{ type: "text", text: "hi" }] }, collect)).toEqual( expect(toSandboxResult({ structuredContent: { x: 1 }, content: [{ type: "text", text: "hi" }] }, collect)).toEqual({
{ x: 1 }, x: 1,
) })
}) })
test("joins text content when no structured content is present", () => { test("joins text content when no structured content is present", () => {
const { collect } = collector() const { collect } = collector()
expect( expect(
toSandboxResult( toSandboxResult(
{ content: [{ type: "text", text: "one" }, { type: "text", text: "two" }] }, {
content: [
{ type: "text", text: "one" },
{ type: "text", text: "two" },
],
},
collect, collect,
), ),
).toBe("one\ntwo") ).toBe("one\ntwo")

View file

@ -2355,7 +2355,13 @@ function Execute(props: ToolProps) {
return ( return (
<> <>
<InlineTool <InlineTool
icon={props.part.state.status === "completed" && !hasRuntimeError() ? "✓" : props.part.state.status === "error" || hasRuntimeError() ? "✗" : "│"} icon={
props.part.state.status === "completed" && !hasRuntimeError()
? "✓"
: props.part.state.status === "error" || hasRuntimeError()
? "✗"
: "│"
}
color={hasRuntimeError() ? theme.error : undefined} color={hasRuntimeError() ? theme.error : undefined}
spinner={isLoading()} spinner={isLoading()}
pending="execute" pending="execute"