fix(core): make V2 reads media-aware and binary-safe (#31038)

This commit is contained in:
Kit Langton 2026-06-05 19:48:34 -04:00 committed by GitHub
commit 83dca45dd5
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
26 changed files with 1709 additions and 120 deletions

View file

@ -509,8 +509,8 @@ describe("Bedrock Converse route", () => {
model,
messages: [
Message.user([
{ type: "media", mediaType: "application/pdf", data: "PDFDATA", filename: "report.pdf" },
{ type: "media", mediaType: "text/csv", data: "CSVDATA" },
{ type: "media", mediaType: "application/pdf", data: "UERGREFUQQ==", filename: "report.pdf" },
{ type: "media", mediaType: "text/csv", data: "Q1NWREFUQQ==" },
]),
],
}),
@ -522,9 +522,9 @@ describe("Bedrock Converse route", () => {
role: "user",
content: [
// Filename round-trips when supplied.
{ document: { format: "pdf", name: "report.pdf", source: { bytes: "PDFDATA" } } },
{ document: { format: "pdf", name: "report.pdf", source: { bytes: "UERGREFUQQ==" } } },
// Falls back to a stable placeholder when filename is missing.
{ document: { format: "csv", name: "document.csv", source: { bytes: "CSVDATA" } } },
{ document: { format: "csv", name: "document.csv", source: { bytes: "Q1NWREFUQQ==" } } },
],
},
],

View file

@ -3,6 +3,7 @@ import { Effect } from "effect"
import { LLM, LLMError, Message, ToolCallPart, Usage } from "../../src"
import { Auth, LLMClient } from "../../src/route"
import * as Gemini from "../../src/protocols/gemini"
import { ProviderShared } from "../../src/protocols/shared"
import { it } from "../lib/effect"
import { fixedResponse } from "../lib/http"
import { sseEvents, sseRaw } from "../lib/sse"
@ -109,6 +110,110 @@ describe("Gemini route", () => {
}),
)
it.effect("continues image tool results as inline vision input without base64 text", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<Gemini.GeminiBody>(
LLM.request({
model,
messages: [
Message.assistant([ToolCallPart.make({ id: "call_image", name: "read", input: { path: "pixel.png" } })]),
Message.tool({
id: "call_image",
name: "read",
result: {
type: "content",
value: [
{ type: "text", text: "Image read successfully" },
{ type: "media", mediaType: "image/png", data: "AAECAw==", filename: "pixel.png" },
],
},
}),
],
}),
)
expect(prepared.body.contents).toEqual([
{ role: "model", parts: [{ functionCall: { name: "read", args: { path: "pixel.png" } } }] },
{
role: "user",
parts: [
{
functionResponse: {
name: "read",
response: { name: "read", content: "Image read successfully" },
},
},
{ inlineData: { mimeType: "image/png", data: "AAECAw==" } },
],
},
])
expect(JSON.stringify(prepared.body.contents)).not.toContain('"content":"AAECAw=="')
}),
)
it.effect("strips matching data URLs to raw base64 inlineData", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<Gemini.GeminiBody>(
LLM.request({
model,
messages: [
Message.user({ type: "media", mediaType: "image/png", data: "data:image/png;base64,AAEC" }),
Message.tool({
id: "call_image",
name: "read",
result: {
type: "content",
value: [{ type: "media", mediaType: "image/jpeg", data: "data:image/jpeg;base64,/9j/" }],
},
}),
],
}),
)
expect(prepared.body.contents).toEqual([
{ role: "user", parts: [{ inlineData: { mimeType: "image/png", data: "AAEC" } }] },
{
role: "user",
parts: [
{ functionResponse: { name: "read", response: { name: "read", content: "" } } },
{ inlineData: { mimeType: "image/jpeg", data: "/9j/" } },
],
},
])
}),
)
for (const [name, media] of [
["mismatched data URL MIME", { mediaType: "image/png", data: "data:image/jpeg;base64,/9j/" }],
["malformed base64", { mediaType: "image/png", data: "%%%=" }],
["unsupported SVG", { mediaType: "image/svg+xml", data: "PHN2Zz4=" }],
] as const)
it.effect(`rejects ${name}`, () =>
Effect.gen(function* () {
const error = yield* LLMClient.prepare(
LLM.request({ model, messages: [Message.user({ type: "media", ...media })] }),
).pipe(Effect.flip)
expect(error.message).toMatch(/does not support|does not match|valid base64/)
}),
)
it.effect("rejects oversized image input", () =>
Effect.gen(function* () {
const error = yield* LLMClient.prepare(
LLM.request({
model,
messages: [
Message.user({
type: "media",
mediaType: "image/png",
data: "A".repeat(ProviderShared.MAX_MEDIA_ENCODED_BYTES + 4),
}),
],
}),
).pipe(Effect.flip)
expect(error.message).toContain("encoded limit")
}),
)
it.effect("omits tools when tool choice is none", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare(

View file

@ -73,7 +73,7 @@ describeRecordedGoldenScenarios([
prefix: "openai-chat",
model: openAIChat,
requires: ["OPENAI_API_KEY"],
scenarios: ["text", "tool-call", "tool-loop"],
scenarios: ["text", "tool-call", "tool-loop", { id: "image-tool-result", maxTokens: 40 }],
},
{
name: "OpenAI Responses gpt-5.5",
@ -123,7 +123,12 @@ describeRecordedGoldenScenarios([
prefix: "gemini",
model: gemini,
requires: ["GOOGLE_GENERATIVE_AI_API_KEY"],
scenarios: [{ id: "text", maxTokens: 80 }, "tool-call", { id: "image", maxTokens: 160 }],
scenarios: [
{ id: "text", maxTokens: 80 },
"tool-call",
{ id: "image", maxTokens: 160 },
{ id: "image-tool-result", maxTokens: 40 },
],
},
{
name: "xAI Grok 3 Mini",

View file

@ -5,6 +5,7 @@ import { LLM, LLMError, Message, Model, ToolCallPart, Usage } from "../../src"
import * as Azure from "../../src/providers/azure"
import * as OpenAI from "../../src/providers/openai"
import * as OpenAIChat from "../../src/protocols/openai-chat"
import { ProviderShared } from "../../src/protocols/shared"
import { Auth, LLMClient } from "../../src/route"
import { it } from "../lib/effect"
import { dynamicResponse, fixedResponse, truncatedStream } from "../lib/http"
@ -223,17 +224,208 @@ describe("OpenAI Chat route", () => {
}),
)
it.effect("rejects unsupported user media content", () =>
it.effect("continues image tool results as vision input without base64 text", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIChat.OpenAIChatBody>(
LLM.request({
model,
messages: [
Message.assistant([ToolCallPart.make({ id: "call_image", name: "read", input: { path: "pixel.png" } })]),
Message.tool({
id: "call_image",
name: "read",
result: {
type: "content",
value: [
{ type: "text", text: "Image read successfully" },
{ type: "media", mediaType: "image/png", data: "AAECAw==", filename: "pixel.png" },
],
},
}),
],
}),
)
expect(prepared.body.messages).toEqual([
{
role: "assistant",
content: null,
tool_calls: [
{
id: "call_image",
type: "function",
function: { name: "read", arguments: encodeJson({ path: "pixel.png" }) },
},
],
},
{ role: "tool", tool_call_id: "call_image", content: "Image read successfully" },
{
role: "user",
content: [{ type: "image_url", image_url: { url: "data:image/png;base64,AAECAw==" } }],
},
])
expect(JSON.stringify(prepared.body.messages)).not.toContain('"content":"AAECAw=="')
}),
)
it.effect("orders parallel tool responses before one aggregated vision message", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIChat.OpenAIChatBody>(
LLM.request({
model,
messages: [
Message.assistant([
ToolCallPart.make({ id: "call_1", name: "read", input: {} }),
ToolCallPart.make({ id: "call_2", name: "read", input: {} }),
]),
Message.make({
role: "tool",
content: [
{
type: "tool-result",
id: "call_1",
name: "read",
result: { type: "content", value: [{ type: "media", mediaType: "image/png", data: "AAEC" }] },
},
{
type: "tool-result",
id: "call_2",
name: "read",
result: { type: "content", value: [{ type: "media", mediaType: "image/jpeg", data: "/9j/" }] },
},
],
}),
],
}),
)
expect(prepared.body.messages.slice(1)).toEqual([
{ role: "tool", tool_call_id: "call_1", content: "" },
{ role: "tool", tool_call_id: "call_2", content: "" },
{
role: "user",
content: [
{ type: "image_url", image_url: { url: "data:image/png;base64,AAEC" } },
{ type: "image_url", image_url: { url: "data:image/jpeg;base64,/9j/" } },
],
},
])
}),
)
it.effect("aggregates consecutive tool images with a following system update", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIChat.OpenAIChatBody>(
LLM.request({
model,
messages: [
Message.tool({
id: "call_1",
name: "read",
result: { type: "content", value: [{ type: "media", mediaType: "image/png", data: "AAEC" }] },
}),
Message.tool({
id: "call_2",
name: "read",
result: { type: "content", value: [{ type: "media", mediaType: "image/webp", data: "UklG" }] },
}),
Message.system("Inspect both images."),
],
}),
)
expect(prepared.body.messages).toEqual([
{ role: "tool", tool_call_id: "call_1", content: "" },
{ role: "tool", tool_call_id: "call_2", content: "" },
{
role: "user",
content: [
{ type: "image_url", image_url: { url: "data:image/png;base64,AAEC" } },
{ type: "image_url", image_url: { url: "data:image/webp;base64,UklG" } },
{ type: "text", text: "<system-update>\nInspect both images.\n</system-update>" },
],
},
])
}),
)
it.effect("appends system updates without replacing multipart user content", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIChat.OpenAIChatBody>(
LLM.request({
model,
messages: [
Message.user({ type: "media", mediaType: "image/png", data: "AAEC" }),
Message.system("Keep the image."),
],
}),
)
expect(prepared.body.messages).toEqual([
{
role: "user",
content: [
{ type: "image_url", image_url: { url: "data:image/png;base64,AAEC" } },
{ type: "text", text: "<system-update>\nKeep the image.\n</system-update>" },
],
},
])
}),
)
for (const [name, media] of [
["mismatched data URL MIME", { mediaType: "image/png", data: "data:image/jpeg;base64,/9j/" }],
["malformed base64", { mediaType: "image/png", data: "not-base64" }],
["unsupported SVG", { mediaType: "image/svg+xml", data: "PHN2Zz4=" }],
] as const)
it.effect(`rejects ${name}`, () =>
Effect.gen(function* () {
const error = yield* LLMClient.prepare(
LLM.request({ model, messages: [Message.user({ type: "media", ...media })] }),
).pipe(Effect.flip)
expect(error.message).toMatch(/does not support|does not match|valid base64/)
}),
)
it.effect("rejects oversized image input", () =>
Effect.gen(function* () {
const error = yield* LLMClient.prepare(
LLM.request({
id: "req_media",
model,
messages: [Message.user({ type: "media", mediaType: "image/png", data: "AAECAw==" })],
messages: [
Message.user({
type: "media",
mediaType: "image/png",
data: "A".repeat(ProviderShared.MAX_MEDIA_ENCODED_BYTES + 4),
}),
],
}),
).pipe(Effect.flip)
expect(error.message).toContain("encoded limit")
}),
)
expect(error.message).toContain("OpenAI Chat user messages only support text content for now")
it.effect("prepares raw and data URL image media as vision input", () =>
Effect.gen(function* () {
const prepared = yield* LLMClient.prepare<OpenAIChat.OpenAIChatBody>(
LLM.request({
id: "req_media",
model,
messages: [
Message.user([
{ type: "media", mediaType: "image/png", data: "AAECAw==" },
{ type: "media", mediaType: "image/jpeg", data: "data:image/jpeg;base64,/9j/" },
]),
],
}),
)
expect(prepared.body.messages).toEqual([
{
role: "user",
content: [
{ type: "image_url", image_url: { url: "data:image/png;base64,AAECAw==" } },
{ type: "image_url", image_url: { url: "data:image/jpeg;base64,/9j/" } },
],
},
])
}),
)

View file

@ -1254,7 +1254,7 @@ describe("OpenAI Responses route", () => {
}),
).pipe(Effect.flip)
expect(error.message).toContain("OpenAI Responses user media content only supports images")
expect(error.message).toContain("OpenAI Responses does not support media type application/pdf")
}),
)

View file

@ -208,6 +208,31 @@ describe("LLMClient tools", () => {
}),
)
it.effect("can retain model media while redacting duplicated structured payloads", () =>
Effect.gen(function* () {
const image = Tool.make({
description: "Return an image.",
parameters: Schema.Struct({}),
success: Schema.Struct({ mime: Schema.String, data: Schema.String }),
execute: () => Effect.succeed({ mime: "image/png", data: "AAECAw==" }),
toStructuredOutput: (output) => ({ mime: output.mime }),
toModelOutput: ({ output }) => [
{ type: "file", source: { type: "data", data: output.data }, mime: output.mime },
],
})
const dispatched = yield* ToolRuntime.dispatch(
{ image },
LLMEvent.toolCall({ id: "call_image", name: "image", input: {} }),
)
expect(dispatched.output).toEqual({
structured: { mime: "image/png" },
content: [{ type: "file", source: { type: "data", data: "AAECAw==" }, mime: "image/png" }],
})
}),
)
it.effect("models canonical tool files with explicit data, url, and file sources", () =>
Effect.sync(() => {
const decode = Schema.decodeUnknownSync(ToolContent)