fix(core): safely recover malformed tool input (#37698)

This commit is contained in:
Kit Langton 2026-07-18 21:52:45 -04:00 committed by GitHub
commit 57ff57595a
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
22 changed files with 876 additions and 93 deletions

View file

@ -1,11 +1,11 @@
import type { LanguageModelV3CallOptions } from "@ai-sdk/provider"
import type { LanguageModelV3, LanguageModelV3CallOptions, LanguageModelV3StreamPart } from "@ai-sdk/provider"
import { AISDK } from "@opencode-ai/core/aisdk"
import { ModelV2 } from "@opencode-ai/core/model"
import { ProviderV2 } from "@opencode-ai/core/provider"
import { LLM, Message } from "@opencode-ai/ai"
import { LLMClient } from "@opencode-ai/ai/route"
import { LLM, LLMError, LLMEvent, Message } from "@opencode-ai/ai"
import { LLMClient, RequestExecutor } from "@opencode-ai/ai/route"
import { expect } from "bun:test"
import { Effect } from "effect"
import { Effect, Layer } from "effect"
import { testEffect } from "./lib/effect"
const it = testEffect(AISDK.locationLayer)
@ -19,6 +19,37 @@ const model = (packageName: string, settings: Record<string, unknown> = {}) =>
limit: { context: 100, output: 20 },
})
const streamModel = (events: ReadonlyArray<LanguageModelV3StreamPart>): LanguageModelV3 => ({
specificationVersion: "v3",
provider: "test",
modelId: "test",
supportedUrls: {},
doGenerate: () => Promise.reject(new Error("Unexpected non-streaming request")),
doStream: () =>
Promise.resolve({
stream: new ReadableStream({
start(controller) {
events.forEach((event) => controller.enqueue(event))
controller.close()
},
}),
}),
})
const usage = {
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
outputTokens: { total: 1, text: 0, reasoning: 0 },
} as const
const client = LLMClient.layer.pipe(
Layer.provide(
Layer.succeed(
RequestExecutor.Service,
RequestExecutor.Service.of({ execute: () => Effect.die("Unexpected HTTP request") }),
),
),
)
it.effect("keys language models by package and flattened overlays", () =>
Effect.gen(function* () {
const aisdk = yield* AISDK.Service
@ -238,3 +269,69 @@ it.effect("projects replay metadata onto AI SDK prompt parts", () =>
])
}),
)
it.effect("emits malformed AI SDK tool input without executing it", () =>
Effect.gen(function* () {
const aisdk = yield* AISDK.Service
const raw = '{"query":"partial'
yield* aisdk.hook.sdk((event) => {
event.sdk = {
languageModel: () =>
streamModel([
{ type: "tool-input-start", id: "call_1", toolName: "lookup" },
{ type: "tool-input-delta", id: "call_1", delta: raw },
{ type: "tool-input-end", id: "call_1" },
{ type: "tool-call", toolCallId: "call_1", toolName: "lookup", input: raw },
{ type: "finish", finishReason: { unified: "tool-calls", raw: "tool_calls" }, usage },
]),
}
})
const resolved = yield* aisdk.model(model("test-ai-sdk"))
const response = yield* LLMClient.generate(LLM.request({ model: resolved, prompt: "Lookup" })).pipe(
Effect.provide(client),
)
expect(response.events.find(LLMEvent.is.toolInputError)).toMatchObject({
id: "call_1",
name: "lookup",
raw,
message: "Invalid JSON input for aisdk tool call lookup",
})
expect(response.events.some(LLMEvent.is.toolInputEnd)).toBeTrue()
expect(response.events.some(LLMEvent.is.toolCall)).toBeFalse()
}),
)
it.effect("keeps malformed provider-executed AI SDK input terminal", () =>
Effect.gen(function* () {
const aisdk = yield* AISDK.Service
const raw = '{"query":"partial'
yield* aisdk.hook.sdk((event) => {
event.sdk = {
languageModel: () =>
streamModel([
{ type: "tool-input-start", id: "call_1", toolName: "web_search", providerExecuted: true },
{ type: "tool-input-delta", id: "call_1", delta: raw },
{ type: "tool-input-end", id: "call_1" },
{
type: "tool-call",
toolCallId: "call_1",
toolName: "web_search",
input: raw,
providerExecuted: true,
},
]),
}
})
const resolved = yield* aisdk.model(model("hosted-test-ai-sdk"))
const error = yield* LLMClient.generate(LLM.request({ model: resolved, prompt: "Search" })).pipe(
Effect.provide(client),
Effect.flip,
)
expect(error).toBeInstanceOf(LLMError)
expect(error.message).toContain("Invalid JSON input for aisdk tool call web_search")
}),
)

View file

@ -565,6 +565,22 @@ Recent work
text: "Partial thought",
state: { itemId: "rs_failed", reasoningEncryptedContent: null },
}),
SessionMessage.AssistantTool.make({
type: "tool",
id: "hosted-completed",
name: "web_search",
executed: true,
providerState: { itemId: "call_completed" },
providerResultState: { itemId: "result_completed" },
state: SessionMessage.ToolStateCompleted.make({
status: "completed",
input: { query: "Effect" },
content: [],
structured: {},
result: { type: "json", value: { found: true } },
}),
time: { created, completed: created },
}),
SessionMessage.AssistantTool.make({
type: "tool",
id: "hosted-failed",
@ -592,6 +608,22 @@ Recent work
expect(messages[0]?.content).toEqual([
{ type: "text", text: "Partial thought" },
{
type: "tool-call",
id: "hosted-completed",
name: "web_search",
input: { query: "Effect" },
providerExecuted: true,
providerMetadata: { provider: { itemId: "call_completed" } },
},
{
type: "tool-result",
id: "hosted-completed",
name: "web_search",
result: { type: "json", value: { found: true } },
providerExecuted: true,
providerMetadata: { provider: { itemId: "result_completed" } },
},
{
type: "tool-call",
id: "hosted-failed",

View file

@ -9,6 +9,8 @@ import { SessionMessage } from "@opencode-ai/core/session/message"
import { SessionV2 } from "@opencode-ai/core/session"
import { ModelV2 } from "@opencode-ai/core/model"
import { ProviderV2 } from "@opencode-ai/core/provider"
import { RelativePath } from "@opencode-ai/core/schema"
import { Snapshot } from "@opencode-ai/core/snapshot"
import { createLLMEventPublisher } from "@opencode-ai/core/session/runner/publish-llm-event"
const sessionID = SessionV2.ID.make("ses_tool_event_test")
@ -229,6 +231,8 @@ test("content-filter finish retains failure evidence until step closeout", async
publisher.publishStepFailure({
cost: Money.USD.make(1.25),
tokens: settlement.tokens,
snapshot: Snapshot.ID.make("tree-end"),
files: [RelativePath.make("src/changed.ts")],
}),
)
expect(published.map((event) => event.type)).toEqual(["session.step.started.1", "session.step.failed.1"])
@ -236,6 +240,8 @@ test("content-filter finish retains failure evidence until step closeout", async
error: { type: "provider.content-filter", message: "Provider blocked the response" },
cost: 1.25,
tokens: { input: 8, output: 2, reasoning: 1 },
snapshot: "tree-end",
files: ["src/changed.ts"],
})
})

View file

@ -4164,6 +4164,299 @@ describe("SessionRunnerLLM", () => {
}),
)
it.effect("continues once after malformed local tool input without exposing raw arguments", () =>
Effect.gen(function* () {
const session = yield* setup
yield* admit(session, "Recover malformed tool input")
const marker = "raw-malformed-marker"
const raw = `{"text":"${marker}`
responses = [
[
LLMEvent.stepStart({ index: 0 }),
LLMEvent.toolInputStart({ id: "call-malformed", name: "echo" }),
LLMEvent.toolInputDelta({ id: "call-malformed", name: "echo", text: raw }),
LLMEvent.toolInputEnd({ id: "call-malformed", name: "echo" }),
LLMEvent.toolInputError({
id: "call-malformed",
name: "echo",
raw,
message: "Invalid JSON input for test tool call echo",
}),
LLMEvent.stepFinish({ index: 0, reason: "tool-calls" }),
LLMEvent.finish({ reason: "tool-calls" }),
],
reply.stop(),
]
yield* session.resume(sessionID)
expect(requests).toHaveLength(2)
expect(executions).toEqual([])
expect(JSON.stringify(requests[1])).not.toContain(marker)
expect(requests[1]?.messages).toEqual(
expect.arrayContaining([
expect.objectContaining({
role: "assistant",
content: expect.arrayContaining([
expect.objectContaining({ type: "tool-call", id: "call-malformed", name: "echo", input: {} }),
]),
}),
expect.objectContaining({
role: "tool",
content: expect.arrayContaining([
expect.objectContaining({
type: "tool-result",
id: "call-malformed",
result: expect.objectContaining({
type: "error",
value: expect.objectContaining({
error: expect.objectContaining({
message: "Tool call arguments were malformed JSON and were not executed. Retry with valid JSON.",
}),
}),
}),
}),
]),
}),
]),
)
const context = yield* session.context(sessionID)
const failed = context.find(
(message): message is SessionMessage.Assistant =>
message.type === "assistant" && message.content.some((item) => item.type === "tool"),
)
expect(failed).toMatchObject({
error: { type: "provider.invalid-output", message: "Invalid JSON input for test tool call echo" },
content: [
{
type: "tool",
id: "call-malformed",
executed: false,
state: {
status: "error",
input: {},
error: {
type: "tool.input-json",
message: "Tool call arguments were malformed JSON and were not executed. Retry with valid JSON.",
},
},
},
],
})
if (!failed) throw new Error("Malformed tool assistant missing")
expect((yield* recordedStepSettlementEvents(sessionID, failed.id)).map((event) => event.type)).toEqual([
"session.step.started.1",
"session.tool.failed.1",
"session.step.failed.1",
])
const database = (yield* Database.Service).db
const durable = yield* database
.select({ type: EventTable.type, data: EventTable.data })
.from(EventTable)
.where(eq(EventTable.aggregate_id, sessionID))
.all()
.pipe(Effect.orDie)
expect(durable.find((event) => event.type === "session.tool.input.ended.1")?.data).toMatchObject({
callID: "call-malformed",
text: raw,
})
}),
)
it.effect("settles a valid sibling before recovering malformed tool input", () =>
Effect.gen(function* () {
const session = yield* setup
yield* admit(session, "Run parallel tools")
toolExecutionGate = yield* Deferred.make<void>()
toolExecutionsStarted = yield* Deferred.make<void>()
toolExecutionsReady = 1
responses = [
[
LLMEvent.stepStart({ index: 0 }),
LLMEvent.toolCall({ id: "call-valid", name: "echo", input: { text: "valid" } }),
LLMEvent.toolInputError({
id: "call-malformed",
name: "echo",
raw: '{"text":"partial',
message: "Invalid JSON input for test tool call echo",
}),
LLMEvent.stepFinish({ index: 0, reason: "tool-calls" }),
LLMEvent.finish({ reason: "tool-calls" }),
],
reply.stop(),
]
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
yield* Deferred.await(toolExecutionsStarted)
expect(requests).toHaveLength(1)
yield* Deferred.succeed(toolExecutionGate, undefined)
yield* Fiber.join(run)
toolExecutionGate = undefined
toolExecutionsStarted = undefined
expect(requests).toHaveLength(2)
expect(executions).toEqual(["valid"])
const request = requests[1]
if (!request) throw new Error("Malformed recovery request missing")
expect(request.messages.flatMap((message) => (message.role === "tool" ? message.content : []))).toEqual(
expect.arrayContaining([
expect.objectContaining({ id: "call-valid", type: "tool-result" }),
expect.objectContaining({ id: "call-malformed", type: "tool-result" }),
]),
)
}),
)
it.effect("does not recover malformed input after sibling execution is interrupted", () =>
Effect.gen(function* () {
const session = yield* setup
yield* admit(session, "Interrupt malformed recovery")
toolExecutionGate = yield* Deferred.make<void>()
toolExecutionsStarted = yield* Deferred.make<void>()
toolExecutionsReady = 1
response = [
LLMEvent.stepStart({ index: 0 }),
LLMEvent.toolCall({ id: "call-valid", name: "echo", input: { text: "blocked" } }),
LLMEvent.toolInputError({
id: "call-malformed",
name: "echo",
raw: '{"text":"partial',
message: "Invalid JSON input for test tool call echo",
}),
LLMEvent.stepFinish({ index: 0, reason: "tool-calls" }),
LLMEvent.finish({ reason: "tool-calls" }),
]
const run = yield* session.resume(sessionID).pipe(Effect.forkChild)
yield* Deferred.await(toolExecutionsStarted)
while (
!(yield* session.context(sessionID)).some(
(message) =>
message.type === "assistant" &&
message.content.some((item) => item.type === "tool" && item.id === "call-malformed"),
)
)
yield* Effect.yieldNow
yield* session.interrupt(sessionID)
toolExecutionGate = undefined
toolExecutionsStarted = undefined
expect(yield* Fiber.await(run)).toMatchObject({ _tag: "Failure" })
expect(requests).toHaveLength(1)
expect(yield* session.context(sessionID)).toMatchObject([
{ type: "user", text: "Interrupt malformed recovery" },
{
type: "assistant",
error: { type: "aborted", message: "Step interrupted" },
content: [
{ type: "tool", id: "call-valid", state: { status: "error", error: { type: "aborted" } } },
{ type: "tool", id: "call-malformed", state: { status: "error" } },
],
},
])
}),
)
it.effect("records malformed provider-executed input as executed", () =>
Effect.gen(function* () {
const session = yield* setup
yield* admit(session, "Fail malformed hosted input")
const failure = new LLMError({
module: "test",
method: "stream",
reason: new InvalidProviderOutputReason({ message: "Invalid hosted tool input" }),
})
responseStream = Stream.fromIterable([
LLMEvent.stepStart({ index: 0 }),
LLMEvent.toolInputStart({ id: "call-hosted", name: "web_search", providerExecuted: true }),
LLMEvent.toolInputDelta({ id: "call-hosted", name: "web_search", text: '{"query":"partial' }),
]).pipe(Stream.concat(Stream.fail(failure)))
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
expect(requireAssistant(yield* session.context(sessionID))).toMatchObject({
error: { type: "provider.invalid-output", message: "Invalid hosted tool input" },
content: [
{
type: "tool",
id: "call-hosted",
executed: true,
state: { status: "error", error: { type: "provider.invalid-output" } },
},
],
})
}),
)
it.effect("replaces malformed input diagnosis with a later provider failure", () =>
Effect.gen(function* () {
const session = yield* setup
yield* admit(session, "Fail after malformed input")
const failure = new LLMError({
module: "test",
method: "stream",
reason: new InvalidProviderOutputReason({ message: "Provider failed after malformed input" }),
})
responseStream = Stream.fromIterable([
LLMEvent.stepStart({ index: 0 }),
LLMEvent.toolInputError({
id: "call-malformed",
name: "echo",
raw: '{"text":"partial',
message: "Invalid JSON input for test tool call echo",
}),
]).pipe(Stream.concat(Stream.fail(failure)))
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toBe(failure)
expect(requireAssistant(yield* session.context(sessionID))).toMatchObject({
error: { type: "provider.invalid-output", message: "Provider failed after malformed input" },
content: [
{
type: "tool",
id: "call-malformed",
executed: false,
state: { status: "error", error: { type: "tool.input-json" } },
},
],
})
expect(requests).toHaveLength(1)
}),
)
it.effect("does not reset the malformed recovery budget after a valid tool step", () =>
Effect.gen(function* () {
const session = yield* setup
yield* admit(session, "Keep producing malformed tools")
const malformed = (id: string) => [
LLMEvent.stepStart({ index: 0 }),
LLMEvent.toolInputError({
id,
name: "echo",
raw: '{"text":"partial',
message: "Invalid JSON input for test tool call echo",
}),
LLMEvent.stepFinish({ index: 0, reason: "tool-calls" }),
LLMEvent.finish({ reason: "tool-calls" }),
]
responses = [
malformed("call-first"),
reply.tool("call-valid-between", "echo", { text: "valid" }),
malformed("call-second"),
reply.stop(),
]
expect(yield* session.resume(sessionID).pipe(Effect.flip)).toMatchObject({
error: {
type: "provider.invalid-output",
message: "Invalid JSON input for test tool call echo",
},
})
expect(requests).toHaveLength(3)
expect(executions).toEqual(["valid"])
expect((yield* recordedEventTypes(sessionID)).filter((type) => type === "session.step.failed.1")).toHaveLength(2)
}),
)
it.effect("does not continue automatically after a provider error follows a local tool call", () =>
Effect.gen(function* () {
const session = yield* setup