From b498a5c6c4be0d5a51eb5c6f2442f33ca169478c Mon Sep 17 00:00:00 2001 From: Aiden Cline <63023139+rekram1-node@users.noreply.github.com> Date: Fri, 31 Jul 2026 12:57:18 -0500 Subject: [PATCH] feat(ai): expand OpenRouter native support (#39907) --- packages/ai/src/cache-policy.ts | 3 +- packages/ai/src/protocols/openai-chat.ts | 145 +++++++++++++++--- packages/ai/src/providers/openrouter.ts | 91 +++++++++-- packages/ai/src/schema/messages.ts | 2 + .../test/provider-options/openrouter.types.ts | 22 +++ packages/ai/test/provider/openrouter.test.ts | 141 ++++++++++++++++- packages/core/src/aisdk-native.ts | 77 +++++++--- packages/core/src/model-resolver.ts | 13 +- packages/core/src/session/generate-node.ts | 5 +- packages/core/src/session/model-request.ts | 5 +- packages/core/test/aisdk-native.test.ts | 36 +++++ packages/core/test/model-resolver.test.ts | 32 ++++ 12 files changed, 507 insertions(+), 65 deletions(-) diff --git a/packages/ai/src/cache-policy.ts b/packages/ai/src/cache-policy.ts index 7679c7cf54..1e314ea68a 100644 --- a/packages/ai/src/cache-policy.ts +++ b/packages/ai/src/cache-policy.ts @@ -38,7 +38,7 @@ const resolve = (policy: CachePolicy | undefined): CachePolicyObject => { // Protocols whose wire format ignores inline cache markers (OpenAI's implicit // prefix caching, Gemini's implicit + out-of-band CachedContent). Skip the // whole policy pass for these — emitting hints would be harmless but pointless. -const RESPECTS_INLINE_HINTS = new Set(["anthropic-messages", "bedrock-converse"]) +const RESPECTS_INLINE_HINTS = new Set(["anthropic-messages", "bedrock-converse", "openrouter"]) const makeHint = (ttlSeconds: number | undefined): CacheHint => ttlSeconds !== undefined ? new CacheHint({ type: "ephemeral", ttlSeconds }) : new CacheHint({ type: "ephemeral" }) @@ -133,6 +133,7 @@ const countHints = (request: LLMRequest) => export const applyCachePolicy = (request: LLMRequest): LLMRequest => { if (!RESPECTS_INLINE_HINTS.has(request.model.route.id)) return request + if (request.model.route.id === "openrouter" && (request.cache === undefined || request.cache === "auto")) return request const policy = resolve(request.cache) if (!policy.tools && !policy.system && !policy.messages) return request diff --git a/packages/ai/src/protocols/openai-chat.ts b/packages/ai/src/protocols/openai-chat.ts index e5793f33b1..e61f39aae2 100644 --- a/packages/ai/src/protocols/openai-chat.ts +++ b/packages/ai/src/protocols/openai-chat.ts @@ -11,6 +11,7 @@ import { Usage, type FinishReason, type FinishReasonDetails, + type CacheHint, type JsonSchema, type LLMRequest, type MediaPart, @@ -38,6 +39,11 @@ export const PATH = "/chat/completions" // The body schema is the provider-native JSON body. `fromRequest` below builds // this shape from the common `LLMRequest`, then `Route.make` validates and // JSON-encodes it before transport. +const OpenAIChatCacheControl = Schema.Struct({ + type: Schema.Literal("ephemeral"), + ttl: Schema.optional(Schema.String), +}) + const OpenAIChatFunction = Schema.Struct({ name: Schema.String, description: Schema.String, @@ -47,6 +53,7 @@ const OpenAIChatFunction = Schema.Struct({ const OpenAIChatTool = Schema.Struct({ type: Schema.tag("function"), function: OpenAIChatFunction, + cache_control: Schema.optional(OpenAIChatCacheControl), }) type OpenAIChatTool = Schema.Schema.Type @@ -61,7 +68,11 @@ const OpenAIChatAssistantToolCall = Schema.Struct({ type OpenAIChatAssistantToolCall = Schema.Schema.Type const OpenAIChatUserContent = Schema.Union([ - Schema.Struct({ type: Schema.Literal("text"), text: Schema.String }), + Schema.Struct({ + type: Schema.Literal("text"), + text: Schema.String, + cache_control: Schema.optional(OpenAIChatCacheControl), + }), Schema.Struct({ type: Schema.Literal("image_url"), image_url: Schema.Struct({ url: Schema.String }), @@ -69,7 +80,10 @@ const OpenAIChatUserContent = Schema.Union([ ]) const OpenAIChatMessage = Schema.Union([ - Schema.Struct({ role: Schema.Literal("system"), content: Schema.String }), + Schema.Struct({ + role: Schema.Literal("system"), + content: Schema.Union([Schema.String, Schema.Array(OpenAIChatUserContent)]), + }), Schema.Struct({ role: Schema.Literal("user"), content: Schema.Union([Schema.String, Schema.Array(OpenAIChatUserContent)]), @@ -83,10 +97,16 @@ const OpenAIChatMessage = Schema.Union([ reasoning: Schema.optional(Schema.String), reasoning_text: Schema.optional(Schema.String), reasoning_details: Schema.optional(Schema.Unknown), + cache_control: Schema.optional(OpenAIChatCacheControl), }), [Schema.Record(Schema.String, Schema.Unknown)], ), - Schema.Struct({ role: Schema.Literal("tool"), tool_call_id: Schema.String, content: Schema.String }), + Schema.Struct({ + role: Schema.Literal("tool"), + tool_call_id: Schema.String, + content: Schema.String, + cache_control: Schema.optional(OpenAIChatCacheControl), + }), ]).pipe(Schema.toTaggedUnion("role")) type OpenAIChatMessage = Schema.Schema.Type @@ -210,13 +230,20 @@ export interface ParserState { // Lowering is the only place that knows how common LLM messages map onto the // OpenAI Chat wire format. Keep provider quirks here instead of leaking native // fields into `LLMRequest`. -const lowerTool = (tool: ToolDefinition, inputSchema: JsonSchema): OpenAIChatTool => ({ +interface LoweringOptions { + readonly cacheControl?: ( + cache: CacheHint | undefined, + ) => Schema.Schema.Type | undefined +} + +const lowerTool = (tool: ToolDefinition, inputSchema: JsonSchema, options: LoweringOptions): OpenAIChatTool => ({ type: "function", function: { name: tool.name, description: tool.description, parameters: ToolSchemaProjection.openAI(inputSchema), }, + cache_control: options.cacheControl?.(tool.cache), }) const lowerToolChoice = (toolChoice: NonNullable) => @@ -258,11 +285,14 @@ const reasoningDetails = (parts: ReadonlyArray, native: unknown) if (isRecord(native) && Array.isArray(native.reasoning_details)) return native.reasoning_details } -const lowerUserMessage = Effect.fn("OpenAIChat.lowerUserMessage")(function* (message: OpenAIChatRequestMessage) { +const lowerUserMessage = Effect.fn("OpenAIChat.lowerUserMessage")(function* ( + message: OpenAIChatRequestMessage, + options: LoweringOptions, +) { const content: Array> = [] for (const part of message.content) { if (part.type === "text") { - content.push({ type: "text", text: part.text }) + content.push({ type: "text", text: part.text, cache_control: options.cacheControl?.(part.cache) }) continue } if (part.type === "media") { @@ -271,14 +301,18 @@ const lowerUserMessage = Effect.fn("OpenAIChat.lowerUserMessage")(function* (mes } return yield* ProviderShared.unsupportedContent("OpenAI Chat", "user", ["text", "media"]) } - if (content.every((part) => part.type === "text")) - return { role: "user" as const, content: content.map((part) => part.text).join("") } + if (content.every((part) => part.type === "text" && part.cache_control === undefined)) + return { + role: "user" as const, + content: content.map((part) => (part.type === "text" ? part.text : "")).join(""), + } return { role: "user" as const, content } }) const lowerAssistantMessage = Effect.fn("OpenAIChat.lowerAssistantMessage")(function* ( message: OpenAIChatRequestMessage, configuredField?: string, + options: LoweringOptions = {}, ) { const content: TextPart[] = [] const reasoning: ReasoningPart[] = [] @@ -316,29 +350,44 @@ const lowerAssistantMessage = Effect.fn("OpenAIChat.lowerAssistantMessage")(func if (reasoning.length === 0) return nativeReasoning return text })() + const cached = message.content.findLast((part) => "cache" in part && part.cache !== undefined) const result = { role: "assistant" as const, content: content.length === 0 ? null : ProviderShared.joinText(content), tool_calls: toolCalls.length === 0 ? undefined : toolCalls, reasoning_details: details, + cache_control: options.cacheControl?.(cached && "cache" in cached ? cached.cache : undefined), } if (field === undefined || reasoningText === undefined) return result return { ...result, [field]: reasoningText } }) -const lowerToolMessages = Effect.fn("OpenAIChat.lowerToolMessages")(function* (message: OpenAIChatRequestMessage) { +const lowerToolMessages = Effect.fn("OpenAIChat.lowerToolMessages")(function* ( + message: OpenAIChatRequestMessage, + options: LoweringOptions, +) { const messages: OpenAIChatMessage[] = [] const images: Array> = [] for (const part of message.content) { if (!ProviderShared.supportsContent(part, ["tool-result"])) return yield* ProviderShared.unsupportedContent("OpenAI Chat", "tool", ["tool-result"]) if (part.result.type !== "content") { - messages.push({ role: "tool", tool_call_id: part.id, content: ProviderShared.toolResultText(part) }) + messages.push({ + role: "tool", + tool_call_id: part.id, + content: ProviderShared.toolResultText(part), + cache_control: options.cacheControl?.(part.cache), + }) continue } const content: ReadonlyArray = part.result.value const text = content.filter((item) => item.type === "text").map((item) => item.text) - messages.push({ role: "tool", tool_call_id: part.id, content: text.join("\n") }) + messages.push({ + role: "tool", + tool_call_id: part.id, + content: text.join("\n"), + cache_control: options.cacheControl?.(part.cache), + }) const files = content.filter((item) => item.type === "file") images.push( ...(yield* Effect.forEach(files, (item) => @@ -352,15 +401,29 @@ const lowerToolMessages = Effect.fn("OpenAIChat.lowerToolMessages")(function* (m const lowerMessage = Effect.fn("OpenAIChat.lowerMessage")(function* ( message: OpenAIChatRequestMessage, reasoningField?: string, + options: LoweringOptions = {}, ) { - if (message.role === "user") return [yield* lowerUserMessage(message)] - if (message.role === "assistant") return [yield* lowerAssistantMessage(message, reasoningField)] - return (yield* lowerToolMessages(message)).messages + if (message.role === "user") return [yield* lowerUserMessage(message, options)] + if (message.role === "assistant") return [yield* lowerAssistantMessage(message, reasoningField, options)] + return (yield* lowerToolMessages(message, options)).messages }) -const lowerMessages = Effect.fn("OpenAIChat.lowerMessages")(function* (request: LLMRequest) { +const lowerMessages = Effect.fn("OpenAIChat.lowerMessages")(function* (request: LLMRequest, options: LoweringOptions) { const system: OpenAIChatMessage[] = - request.system.length === 0 ? [] : [{ role: "system", content: ProviderShared.joinText(request.system) }] + request.system.length === 0 + ? [] + : request.system.some((part) => part.cache !== undefined) && options.cacheControl !== undefined + ? [ + { + role: "system", + content: request.system.map((part) => ({ + type: "text", + text: part.text, + cache_control: options.cacheControl?.(part.cache), + })), + }, + ] + : [{ role: "system", content: ProviderShared.joinText(request.system) }] const messages = [...system] const pendingImages: Array> = [] const flushImages = () => { @@ -371,28 +434,53 @@ const lowerMessages = Effect.fn("OpenAIChat.lowerMessages")(function* (request: if (message.role === "system") { const part = yield* ProviderShared.wrappedSystemUpdate("OpenAI Chat", message) if (pendingImages.length > 0) { - messages.push({ role: "user", content: [...pendingImages.splice(0), { type: "text", text: part.text }] }) + messages.push({ + role: "user", + content: [ + ...pendingImages.splice(0), + { type: "text", text: part.text, cache_control: options.cacheControl?.(part.cache) }, + ], + }) continue } const previous = messages.at(-1) if (previous?.role === "user" && typeof previous.content === "string") - messages[messages.length - 1] = { role: "user", content: `${previous.content}\n${part.text}` } + messages[messages.length - 1] = options.cacheControl?.(part.cache) + ? { + role: "user", + content: [ + { type: "text", text: previous.content }, + { type: "text", text: part.text, cache_control: options.cacheControl(part.cache) }, + ], + } + : { role: "user", content: `${previous.content}\n${part.text}` } else if (previous?.role === "user" && Array.isArray(previous.content)) messages[messages.length - 1] = { role: "user", - content: [...previous.content, { type: "text", text: part.text }], + content: [ + ...previous.content, + { type: "text", text: part.text, cache_control: options.cacheControl?.(part.cache) }, + ], } - else messages.push({ role: "user", content: part.text }) + else + messages.push( + options.cacheControl?.(part.cache) + ? { + role: "user", + content: [{ type: "text", text: part.text, cache_control: options.cacheControl(part.cache) }], + } + : { role: "user", content: part.text }, + ) continue } if (message.role === "tool") { - const lowered = yield* lowerToolMessages(message) + const lowered = yield* lowerToolMessages(message, options) messages.push(...lowered.messages) pendingImages.push(...lowered.images) continue } flushImages() - messages.push(...(yield* lowerMessage(message, request.model.compatibility?.reasoningField))) + messages.push(...(yield* lowerMessage(message, request.model.compatibility?.reasoningField, options))) } flushImages() return messages @@ -406,7 +494,10 @@ const lowerOptions = (request: LLMRequest) => { } } -const fromRequest = Effect.fn("OpenAIChat.fromRequest")(function* (request: LLMRequest) { +export const fromRequest = Effect.fn("OpenAIChat.fromRequest")(function* ( + request: LLMRequest, + options: LoweringOptions = {}, +) { // `fromRequest` returns the provider body only. Endpoint, auth, framing, // validation, and HTTP execution are composed by `Route.make`. const reasoningField = request.model.compatibility?.reasoningField @@ -419,12 +510,16 @@ const fromRequest = Effect.fn("OpenAIChat.fromRequest")(function* (request: LLMR const maxTokensField = request.model.compatibility?.maxTokensField ?? "max_tokens" return { model: request.model.id, - messages: yield* lowerMessages(request), + messages: yield* lowerMessages(request, options), tools: request.tools.length === 0 ? undefined : request.tools.map((tool) => - lowerTool(tool, ToolSchemaProjection.modelCompatibility(tool.inputSchema, toolSchemaCompatibility)), + lowerTool( + tool, + ToolSchemaProjection.modelCompatibility(tool.inputSchema, toolSchemaCompatibility), + options, + ), ), tool_choice: request.toolChoice ? yield* lowerToolChoice(request.toolChoice) : undefined, stream: true as const, diff --git a/packages/ai/src/providers/openrouter.ts b/packages/ai/src/providers/openrouter.ts index 29f75b409e..a3681b096f 100644 --- a/packages/ai/src/providers/openrouter.ts +++ b/packages/ai/src/providers/openrouter.ts @@ -4,21 +4,72 @@ import { Endpoint } from "../route/endpoint" import { Framing } from "../route/framing" import { Protocol } from "../route/protocol" import { AuthOptions, type ProviderAuthOption } from "../route/auth-options" -import { ProviderID, type ModelID, type ProviderOptions } from "../schema" +import { ProviderID, type CacheHint, type ModelID, type ProviderOptions } from "../schema" import type { ProviderPackage } from "../provider-package" import * as OpenAICompatibleProfiles from "./openai-compatible-profile" import * as OpenAIChat from "../protocols/openai-chat" +import { newBreakpoints, ttlBucket } from "../protocols/utils/cache" import { isRecord } from "../protocols/shared" export const profile = OpenAICompatibleProfiles.profiles.openrouter export const id = ProviderID.make(profile.provider) const ADAPTER = "openrouter" +type OpenRouterString = Known | (string & {}) + +export interface OpenRouterProviderRouting { + readonly [key: string]: unknown + readonly order?: ReadonlyArray + readonly allow_fallbacks?: boolean + readonly require_parameters?: boolean + readonly data_collection?: OpenRouterString<"allow" | "deny"> + readonly only?: ReadonlyArray + readonly ignore?: ReadonlyArray + readonly quantizations?: ReadonlyArray + readonly sort?: OpenRouterString<"price" | "throughput" | "latency"> + readonly max_price?: Readonly<{ + prompt?: number | string + completion?: number | string + image?: number | string + audio?: number | string + request?: number | string + }> + readonly zdr?: boolean +} + +export type OpenRouterPlugin = + | Readonly<{ + id: "web" + max_results?: number + search_prompt?: string + engine?: OpenRouterString<"native" | "exa"> + }> + | Readonly<{ id: "file-parser"; max_files?: number; pdf?: { engine?: string } }> + | Readonly<{ id: "moderation" }> + | Readonly<{ id: "response-healing" }> + | Readonly<{ id: "auto-router"; allowed_models?: ReadonlyArray }> + | Readonly<{ id: string & {}; [key: string]: unknown }> + export interface OpenRouterOptions { readonly [key: string]: unknown - readonly usage?: boolean | Record - readonly reasoning?: Record + readonly debug?: Readonly<{ echo_upstream_body?: boolean }> + readonly models?: ReadonlyArray + readonly plugins?: ReadonlyArray readonly promptCacheKey?: string + readonly provider?: OpenRouterProviderRouting + readonly reasoning?: Readonly<{ + enabled?: boolean + exclude?: boolean + effort?: OpenRouterString<"none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"> + max_tokens?: number + }> + readonly usage?: boolean | Readonly<{ include: boolean }> + readonly user?: string + readonly web_search_options?: Readonly<{ + max_results?: number + search_prompt?: string + engine?: OpenRouterString<"native" | "exa"> + }> } export type OpenRouterProviderOptionsInput = ProviderOptions & { @@ -47,7 +98,7 @@ export const protocol = Protocol.make({ body: { schema: OpenRouterBody, from: (request) => - OpenAIChat.protocol.body.from(request).pipe( + OpenAIChat.fromRequest(request, { cacheControl: cacheControl() }).pipe( Effect.map((body) => { const sourceAssistants = request.messages.filter((message) => message.role === "assistant") let assistantIndex = 0 @@ -78,16 +129,38 @@ export const protocol = Protocol.make({ stream: OpenAIChat.protocol.stream, }) +const cacheControl = () => { + const breakpoints = newBreakpoints(4) + return (cache: CacheHint | undefined) => { + if (cache === undefined || breakpoints.remaining === 0) return undefined + breakpoints.remaining -= 1 + return { + type: "ephemeral" as const, + ...(ttlBucket(cache.ttlSeconds) === "1h" ? { ttl: "1h" } : {}), + } + } +} + const bodyOptions = (input: unknown) => { const openrouter = isRecord(input) ? input : {} + const { promptCacheKey, ...options } = openrouter return { - ...(openrouter.usage === true + ...options, + ...(openrouter.usage === undefined || openrouter.usage === true ? { usage: { include: true } } - : isRecord(openrouter.usage) - ? { usage: openrouter.usage } - : {}), + : openrouter.usage === false + ? { usage: { include: false } } + : isRecord(openrouter.usage) + ? { usage: openrouter.usage } + : {}), + ...(Array.isArray(openrouter.models) ? { models: openrouter.models } : {}), + ...(isRecord(openrouter.provider) ? { provider: openrouter.provider } : {}), + ...(Array.isArray(openrouter.plugins) ? { plugins: openrouter.plugins } : {}), + ...(isRecord(openrouter.web_search_options) ? { web_search_options: openrouter.web_search_options } : {}), + ...(isRecord(openrouter.debug) ? { debug: openrouter.debug } : {}), + ...(typeof openrouter.user === "string" ? { user: openrouter.user } : {}), ...(isRecord(openrouter.reasoning) ? { reasoning: openrouter.reasoning } : {}), - ...(typeof openrouter.promptCacheKey === "string" ? { prompt_cache_key: openrouter.promptCacheKey } : {}), + ...(typeof promptCacheKey === "string" ? { prompt_cache_key: promptCacheKey } : {}), } } diff --git a/packages/ai/src/schema/messages.ts b/packages/ai/src/schema/messages.ts index bf859f946c..d3689dc0e3 100644 --- a/packages/ai/src/schema/messages.ts +++ b/packages/ai/src/schema/messages.ts @@ -124,6 +124,7 @@ export const ToolCallPart = Object.assign( name: Schema.String, input: Schema.Unknown, providerExecuted: Schema.optional(Schema.Boolean), + cache: Schema.optional(CacheHint), metadata: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)), providerMetadata: Schema.optional(ProviderMetadata), }).annotate({ identifier: "LLM.Content.ToolCall" }), @@ -168,6 +169,7 @@ export const ReasoningPart = Schema.Struct({ type: Schema.Literal("reasoning"), text: Schema.String, encrypted: Schema.optional(Schema.String), + cache: Schema.optional(CacheHint), metadata: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)), providerMetadata: Schema.optional(ProviderMetadata), }).annotate({ identifier: "LLM.Content.Reasoning" }) diff --git a/packages/ai/test/provider-options/openrouter.types.ts b/packages/ai/test/provider-options/openrouter.types.ts index 8982a50edb..e9c30a1e8e 100644 --- a/packages/ai/test/provider-options/openrouter.types.ts +++ b/packages/ai/test/provider-options/openrouter.types.ts @@ -5,6 +5,28 @@ const model = OpenRouter.provider.model("anthropic/claude-sonnet-4.5") LLM.request({ model, prompt: "Hello", providerOptions: { openrouter: { usage: true } } }) +LLM.request({ + model, + prompt: "Hello", + providerOptions: { + openrouter: { + models: ["google/gemini-3.1-pro"], + provider: { + order: ["anthropic"], + require_parameters: true, + data_collection: "future-policy", + sort: "future-sort", + max_price: { prompt: "0.50" }, + }, + reasoning: { effort: "future-effort", exclude: false }, + plugins: [{ id: "future-plugin", enabled: true }], + web_search_options: { engine: "future-engine" }, + debug: { echo_upstream_body: true }, + user: "user_123", + }, + }, +}) + LLM.request({ model, prompt: "Hello", diff --git a/packages/ai/test/provider/openrouter.test.ts b/packages/ai/test/provider/openrouter.test.ts index e2a61c250e..1121631b3e 100644 --- a/packages/ai/test/provider/openrouter.test.ts +++ b/packages/ai/test/provider/openrouter.test.ts @@ -1,6 +1,6 @@ import { describe, expect } from "bun:test" import { Effect } from "effect" -import { LLM, Message } from "../../src" +import { CacheHint, LLM, Message } from "../../src" import { LLMClient } from "../../src/route" import { compileRequest } from "../../src/route/client" import * as OpenRouter from "../../src/providers/openrouter" @@ -27,10 +27,131 @@ describe("OpenRouter", () => { model: "openai/gpt-4o-mini", messages: [{ role: "user", content: "Say hello." }], stream: true, + usage: { include: true }, }) }), ) + it.effect("lowers the native cache policy to OpenRouter cache controls", () => + Effect.gen(function* () { + const prepared = yield* compileRequest( + LLM.request({ + model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + system: [ + { type: "text", text: "Base agent", cache: new CacheHint({ type: "ephemeral", ttlSeconds: 3_600 }) }, + { type: "text", text: "Project instructions" }, + ], + tools: [{ name: "lookup", description: "Lookup", inputSchema: { type: "object", properties: {} } }], + prompt: "Hello", + cache: { tools: true, system: true, messages: { tail: 1 } }, + }), + ) + + expect(prepared.body).toMatchObject({ + tools: [{ cache_control: { type: "ephemeral" } }], + messages: [ + { + role: "system", + content: [ + { text: "Base agent", cache_control: { type: "ephemeral", ttl: "1h" } }, + { text: "Project instructions", cache_control: { type: "ephemeral" } }, + ], + }, + { + role: "user", + content: [{ text: "Hello", cache_control: { type: "ephemeral" } }], + }, + ], + }) + }), + ) + + it.effect("lowers manual assistant and tool-result cache hints", () => + Effect.gen(function* () { + const prepared = yield* compileRequest( + LLM.request({ + model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: "none", + messages: [ + Message.user("Call the tool"), + Message.assistant([ + { type: "text", text: "Calling", cache: new CacheHint({ type: "ephemeral" }) }, + { type: "tool-call", id: "call_1", name: "lookup", input: {} }, + ]), + Message.tool({ + id: "call_1", + name: "lookup", + result: "Done", + cache: new CacheHint({ type: "ephemeral", ttlSeconds: 3_600 }), + }), + ], + }), + ) + + expect(prepared.body.messages).toMatchObject([ + { role: "user", content: "Call the tool" }, + { role: "assistant", content: "Calling", cache_control: { type: "ephemeral" } }, + { role: "tool", content: '"Done"', cache_control: { type: "ephemeral", ttl: "1h" } }, + ]) + }), + ) + + it.effect("caps manual cache controls at four breakpoints", () => + Effect.gen(function* () { + const cache = new CacheHint({ type: "ephemeral" }) + const prepared = yield* compileRequest( + LLM.request({ + model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: "none", + system: [1, 2, 3, 4, 5].map((index) => ({ type: "text" as const, text: `System ${index}`, cache })), + prompt: "Hello", + }), + ) + + const system = prepared.body.messages[0] + expect(system?.role).toBe("system") + expect( + system && Array.isArray(system.content) + ? system.content.filter((part) => "cache_control" in part && part.cache_control !== undefined) + : [], + ).toHaveLength(4) + }), + ) + + it.effect("preserves cache policy hints on reasoning-only assistant messages", () => + Effect.gen(function* () { + const prepared = yield* compileRequest( + LLM.request({ + model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: { messages: "latest-assistant" }, + messages: [Message.user("Think"), Message.assistant([{ type: "reasoning", text: "Reasoning" }])], + }), + ) + + expect(prepared.body.messages).toMatchObject([ + { role: "user", content: "Think" }, + { role: "assistant", cache_control: { type: "ephemeral" } }, + ]) + }), + ) + + it.effect("allows usage accounting to be disabled explicitly", () => + Effect.gen(function* () { + const prepared = yield* compileRequest( + LLM.request({ + model: OpenRouter.configure({ + apiKey: "test-key", + providerOptions: { openrouter: { usage: false } }, + }).model("openai/gpt-4o-mini"), + cache: "none", + prompt: "Hello", + }), + ) + + expect(prepared.body.usage).toEqual({ include: false }) + }), + ) + it.effect("applies OpenRouter payload options from the model helper", () => Effect.gen(function* () { const prepared = yield* compileRequest( @@ -42,6 +163,13 @@ describe("OpenRouter", () => { usage: true, reasoning: { effort: "high" }, promptCacheKey: "session_123", + models: ["anthropic/claude-sonnet-4.6", "google/gemini-3.1-pro"], + provider: { order: ["anthropic", "google"], require_parameters: true }, + plugins: [{ id: "response-healing" }], + web_search_options: { engine: "native", max_results: 3 }, + debug: { echo_upstream_body: true }, + user: "user_123", + future_option: { enabled: true }, }, }, }).model("anthropic/claude-3.7-sonnet:thinking"), @@ -53,6 +181,13 @@ describe("OpenRouter", () => { usage: { include: true }, reasoning: { effort: "high" }, prompt_cache_key: "session_123", + models: ["anthropic/claude-sonnet-4.6", "google/gemini-3.1-pro"], + provider: { order: ["anthropic", "google"], require_parameters: true }, + plugins: [{ id: "response-healing" }], + web_search_options: { engine: "native", max_results: 3 }, + debug: { echo_upstream_body: true }, + user: "user_123", + future_option: { enabled: true }, }) }), ) @@ -104,6 +239,7 @@ describe("OpenRouter", () => { const prepared = yield* compileRequest( LLM.request({ model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: "none", messages: [ Message.assistant([ { @@ -137,6 +273,7 @@ describe("OpenRouter", () => { const prepared = yield* compileRequest( LLM.request({ model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: "none", messages: [ Message.assistant({ type: "reasoning", @@ -162,6 +299,7 @@ describe("OpenRouter", () => { const prepared = yield* compileRequest( LLM.request({ model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: "none", messages: [ Message.assistant({ type: "reasoning", @@ -183,6 +321,7 @@ describe("OpenRouter", () => { const prepared = yield* compileRequest( LLM.request({ model: OpenRouter.configure({ apiKey: "test-key" }).model("anthropic/claude-sonnet-4.6"), + cache: "none", messages: [Message.assistant({ type: "reasoning", text: "Thinking" })], }), ) diff --git a/packages/core/src/aisdk-native.ts b/packages/core/src/aisdk-native.ts index 675e1d8ad9..0d888bfc1b 100644 --- a/packages/core/src/aisdk-native.ts +++ b/packages/core/src/aisdk-native.ts @@ -1,10 +1,13 @@ export * as AISDKNative from "./aisdk-native" import { isRecord } from "@opencode-ai/ai/utils/record" +import { Provider } from "./provider" export interface Mapping { readonly package: string readonly settings: Readonly> + readonly headers?: Readonly> + readonly body?: Readonly> } export function map(packageName: string | undefined, settings: Readonly>): Mapping | undefined { @@ -20,14 +23,7 @@ export function map(packageName: string | undefined, settings: Readonly>) { return { providerOptions: { gemini: options } } } +function mapOpenRouter( + settings: Readonly>, + baseSettings: Readonly>, +): Mapping { + const headers = + Provider.mergeHeaders( + { + ...(typeof settings.appName === "string" ? { "X-OpenRouter-Title": settings.appName } : {}), + ...(typeof settings.appUrl === "string" ? { "HTTP-Referer": settings.appUrl } : {}), + ...(isStringRecord(settings.api_keys) && Object.keys(settings.api_keys).length > 0 + ? { "X-Provider-API-Keys": JSON.stringify(settings.api_keys) } + : {}), + }, + isStringRecord(settings.headers) ? settings.headers : undefined, + ) ?? {} + return { + package: "@opencode-ai/ai/providers/openrouter", + settings: { + ...baseSettings, + ...mapAPIKey(settings), + ...mapOpenRouterOptions(settings), + }, + ...(Object.keys(headers).length > 0 ? { headers } : {}), + ...(isRecord(settings.extraBody) ? { body: settings.extraBody } : {}), + } +} + +function mapOpenRouterOptions(settings: Readonly>) { + const options = Object.fromEntries( + Object.entries(settings).filter( + ([key]) => + ![ + "apiKey", + "api_keys", + "appName", + "appUrl", + "authToken", + "baseURL", + "chunkTimeout", + "compatibility", + "extraBody", + "fetch", + "headers", + "timeout", + ].includes(key), + ), + ) + if (Object.keys(options).length === 0) return {} + return { providerOptions: { openrouter: options } } +} + +function isStringRecord(value: unknown): value is Readonly> { + return isRecord(value) && Object.values(value).every((item) => typeof item === "string") +} + function mapXAIOptions(settings: Readonly>) { const options = { ...(typeof settings.reasoningEffort === "string" ? { reasoningEffort: settings.reasoningEffort } : {}), @@ -78,13 +129,3 @@ function mapXAIOptions(settings: Readonly>) { if (Object.keys(options).length === 0) return {} return { providerOptions: { xai: options } } } - -function mapProviderOptions(namespace: string, settings: Readonly>) { - const values = Object.fromEntries( - Object.entries(settings).filter( - ([key]) => !["apiKey", "authToken", "baseURL", "chunkTimeout", "fetch", "timeout"].includes(key), - ), - ) - if (Object.keys(values).length === 0) return {} - return { providerOptions: { [namespace]: values } } -} diff --git a/packages/core/src/model-resolver.ts b/packages/core/src/model-resolver.ts index d397c5816d..bcfa5aae50 100644 --- a/packages/core/src/model-resolver.ts +++ b/packages/core/src/model-resolver.ts @@ -96,9 +96,7 @@ const providerHeaders = (model: Info) => { return Provider.mergeHeaders(generated.size === 0 ? undefined : Object.fromEntries(generated), model.headers) } -const providerOptions = ( - model: Info, -): { readonly [key: string]: { readonly [key: string]: unknown } } | undefined => { +const providerOptions = (model: Info): { readonly [key: string]: { readonly [key: string]: unknown } } | undefined => { if (!Provider.isAISDK(model.package) || model.settings === undefined) return undefined const { apiKey: _, baseURL: _baseURL, ...settings } = model.settings if (Object.keys(settings).length === 0) return undefined @@ -202,8 +200,8 @@ export const fromCatalogModel = ( const settings = { ...(credential ? withoutNativeAuthSettings(mapped) : mapped), ...nativeCredentialSettings(specifier, credential), - headers: resolved.headers, - body: resolved.body, + headers: Provider.mergeHeaders(mapping?.headers, resolved.headers), + body: Provider.mergeOverlay(mapping?.body, resolved.body), limits: { context: resolved.limit.context, input: resolved.limit.input, output: resolved.limit.output }, } return yield* Effect.try({ @@ -266,10 +264,7 @@ export const layer = Layer.effect( const integrations = yield* Integration.Service const npm = yield* Npm.Service const aisdk = yield* AISDK.Service - const load = Effect.fn("ModelResolver.resolveModel")(function* ( - selected: Info, - variant?: VariantID, - ) { + const load = Effect.fn("ModelResolver.resolveModel")(function* (selected: Info, variant?: VariantID) { const provider = yield* catalog.provider.get(selected.providerID) const connection = yield* integrations.connection.active( provider?.integrationID ?? Integration.ID.make(selected.providerID), diff --git a/packages/core/src/session/generate-node.ts b/packages/core/src/session/generate-node.ts index 9a5c1031fd..35943097d2 100644 --- a/packages/core/src/session/generate-node.ts +++ b/packages/core/src/session/generate-node.ts @@ -71,7 +71,10 @@ export const layer = Layer.effect( LLM.request({ model: model.model, http: { headers: SessionModelHeaders.make(selection.session, app) }, - providerOptions: { openai: { promptCacheKey } }, + providerOptions: + model.model.route.id === "openrouter" + ? { openrouter: { promptCacheKey } } + : { openai: { promptCacheKey } }, system: contextEvent.system, messages: contextEvent.messages, tools: hookedTools, diff --git a/packages/core/src/session/model-request.ts b/packages/core/src/session/model-request.ts index 18ed3a1933..ba87995b63 100644 --- a/packages/core/src/session/model-request.ts +++ b/packages/core/src/session/model-request.ts @@ -158,7 +158,10 @@ export const layer = Layer.effect( http: { headers: SessionModelHeaders.make(session, app), }, - providerOptions: { openai: { promptCacheKey } }, + providerOptions: + model.route.id === "openrouter" + ? { openrouter: { promptCacheKey } } + : { openai: { promptCacheKey } }, system: contextEvent.system, messages: unsupportedParts(contextEvent.messages, resolved.capabilities), tools: hookedTools, diff --git a/packages/core/test/aisdk-native.test.ts b/packages/core/test/aisdk-native.test.ts index bd04920fe9..6e243852af 100644 --- a/packages/core/test/aisdk-native.test.ts +++ b/packages/core/test/aisdk-native.test.ts @@ -2,6 +2,42 @@ import { describe, expect, test } from "bun:test" import { AISDKNative } from "@opencode-ai/core/aisdk-native" describe("AISDKNative", () => { + test("maps OpenRouter settings to native destinations", () => { + expect( + AISDKNative.map("@openrouter/ai-sdk-provider", { + appName: "OpenCode", + appUrl: "https://opencode.ai", + headers: { "x-openrouter-title": "Configured", "x-provider-api-keys": "Configured BYOK" }, + api_keys: { anthropic: "provider-key" }, + extraBody: { transforms: ["middle-out"] }, + models: ["anthropic/claude-sonnet-4.6"], + provider: { only: ["anthropic"], require_parameters: true }, + reasoning: { effort: "high" }, + promptCacheKey: "session_123", + future_option: { enabled: true }, + }), + ).toEqual({ + package: "@opencode-ai/ai/providers/openrouter", + settings: { + providerOptions: { + openrouter: { + models: ["anthropic/claude-sonnet-4.6"], + provider: { only: ["anthropic"], require_parameters: true }, + reasoning: { effort: "high" }, + promptCacheKey: "session_123", + future_option: { enabled: true }, + }, + }, + }, + headers: { + "x-openrouter-title": "Configured", + "HTTP-Referer": "https://opencode.ai", + "x-provider-api-keys": "Configured BYOK", + }, + body: { transforms: ["middle-out"] }, + }) + }) + test("maps every Google thinking setting", () => { expect( AISDKNative.map("@ai-sdk/google", { diff --git a/packages/core/test/model-resolver.test.ts b/packages/core/test/model-resolver.test.ts index b7838b8c9f..55c6411232 100644 --- a/packages/core/test/model-resolver.test.ts +++ b/packages/core/test/model-resolver.test.ts @@ -1,5 +1,6 @@ import { describe, expect } from "bun:test" import { LLM, Model } from "@opencode-ai/ai" +import { OpenAIChat } from "@opencode-ai/ai/protocols" import { compileRequest } from "@opencode-ai/ai/route/client" import { Effect } from "effect" import { Headers } from "effect/unstable/http" @@ -546,6 +547,37 @@ describe("ModelResolver", () => { }), ) + it.effect("merges mapped OpenRouter headers and body with catalog overlays", () => + ModelResolver.fromCatalogModel( + model(Provider.aisdk("@openrouter/ai-sdk-provider"), { + settings: { + appName: "OpenCode", + appUrl: "https://opencode.ai", + extraBody: { transforms: ["middle-out"], provider: { sort: "price" } }, + }, + headers: { "X-OpenRouter-Title": "Custom" }, + body: { provider: { only: ["anthropic"] } }, + }), + undefined, + { + loadPackage: () => + Effect.succeed({ + model: (modelID, settings) => { + expect(settings.headers).toEqual({ + "HTTP-Referer": "https://opencode.ai", + "X-OpenRouter-Title": "Custom", + }) + expect(settings.body).toEqual({ + transforms: ["middle-out"], + provider: { sort: "price", only: ["anthropic"] }, + }) + return Model.make({ id: modelID, provider: "openrouter", route: OpenAIChat.route }) + }, + }), + }, + ), + ) + it.effect("loads supported AISDK catalog packages as native routes", () => Effect.gen(function* () { const google = yield* ModelResolver.fromCatalogModel(