refactor(ai): unify prompt cache configuration
This commit is contained in:
parent
f6ea6b1762
commit
a214ac39de
19 changed files with 64 additions and 59 deletions
|
|
@ -10,16 +10,16 @@
|
|||
//
|
||||
// Manual `cache: CacheHint` placements on individual parts are preserved and
|
||||
// count against the four-breakpoint budget; auto only fills remaining slots.
|
||||
import { CacheHint, type CachePolicy, type CachePolicyObject } from "./schema/options"
|
||||
import { CacheHint, type CacheExplicit, type CachePolicy } from "./schema/options"
|
||||
import { LLMRequest, Message, ToolDefinition, type ContentPart } from "./schema/messages"
|
||||
|
||||
const AUTO: CachePolicyObject = {
|
||||
const AUTO: Omit<CacheExplicit, "mode" | "key"> = {
|
||||
tools: true,
|
||||
system: true,
|
||||
messages: { tail: 1 },
|
||||
}
|
||||
|
||||
const NONE: CachePolicyObject = {}
|
||||
const NONE: Omit<CacheExplicit, "mode" | "key"> = {}
|
||||
const BREAKPOINT_CAP = 4
|
||||
|
||||
// Resolution rules:
|
||||
|
|
@ -29,8 +29,8 @@ const BREAKPOINT_CAP = 4
|
|||
// - "auto" → tools + first/last system + final message boundary.
|
||||
// - "none" → no auto placement; manual `CacheHint`s still flow.
|
||||
// - object form → exactly what the caller asked for.
|
||||
const resolve = (policy: CachePolicy | undefined): CachePolicyObject => {
|
||||
if (policy === undefined || policy === "auto") return AUTO
|
||||
const resolve = (policy: CachePolicy | undefined): Omit<CacheExplicit, "mode" | "key"> => {
|
||||
if (policy === undefined || (policy !== "none" && policy.mode === "auto")) return AUTO
|
||||
if (policy === "none") return NONE
|
||||
return policy
|
||||
}
|
||||
|
|
@ -103,7 +103,7 @@ const markMessageAt = (
|
|||
|
||||
const markMessages = (
|
||||
messages: ReadonlyArray<Message>,
|
||||
strategy: NonNullable<CachePolicyObject["messages"]>,
|
||||
strategy: NonNullable<CacheExplicit["messages"]>,
|
||||
hint: CacheHint,
|
||||
budget: Budget,
|
||||
): ReadonlyArray<Message> => {
|
||||
|
|
@ -133,7 +133,11 @@ const countHints = (request: LLMRequest) =>
|
|||
|
||||
export const applyCachePolicy = (request: LLMRequest): LLMRequest => {
|
||||
if (!RESPECTS_INLINE_HINTS.has(request.model.route.id)) return request
|
||||
if (request.model.route.id === "openrouter" && (request.cache === undefined || request.cache === "auto")) return request
|
||||
if (
|
||||
request.model.route.id === "openrouter" &&
|
||||
(request.cache === undefined || (request.cache !== "none" && request.cache.mode === "auto"))
|
||||
)
|
||||
return request
|
||||
const policy = resolve(request.cache)
|
||||
if (!policy.tools && !policy.system && !policy.messages) return request
|
||||
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import type { Content } from "@opencode-ai/schema/tool"
|
|||
import { HttpTransport } from "../route/transport"
|
||||
import { Protocol } from "../route/protocol"
|
||||
import {
|
||||
cacheKey,
|
||||
LLMError,
|
||||
LLMEvent,
|
||||
Usage,
|
||||
|
|
@ -542,7 +543,7 @@ const lowerOptions = (request: LLMRequest) => {
|
|||
return {
|
||||
...(options.instructions ? { instructions: options.instructions } : {}),
|
||||
...(options.store !== undefined ? { store: options.store } : {}),
|
||||
...(request.promptCacheKey ? { prompt_cache_key: request.promptCacheKey } : {}),
|
||||
...(cacheKey(request.cache) ? { prompt_cache_key: cacheKey(request.cache) } : {}),
|
||||
...(options.include ? { include: options.include } : {}),
|
||||
...(options.reasoningEffort || options.reasoningSummary
|
||||
? { reasoning: { effort: options.reasoningEffort, summary: options.reasoningSummary } }
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ import { Endpoint } from "../route/endpoint"
|
|||
import { Framing } from "../route/framing"
|
||||
import { Protocol } from "../route/protocol"
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options"
|
||||
import { ProviderID, type CacheHint, type ModelID, type ProviderOptions } from "../schema"
|
||||
import { cacheKey, ProviderID, type CacheHint, type ModelID, type ProviderOptions } from "../schema"
|
||||
import type { ProviderPackage } from "../provider-package"
|
||||
import * as OpenAICompatibleProfiles from "./openai-compatible-profile"
|
||||
import * as OpenAIChat from "../protocols/openai-chat"
|
||||
|
|
@ -121,7 +121,7 @@ export const protocol = Protocol.make({
|
|||
...body,
|
||||
messages,
|
||||
...bodyOptions(request.providerOptions?.openrouter),
|
||||
...(request.promptCacheKey ? { prompt_cache_key: request.promptCacheKey } : {}),
|
||||
...(cacheKey(request.cache) ? { prompt_cache_key: cacheKey(request.cache) } : {}),
|
||||
} as OpenRouterBody
|
||||
}),
|
||||
),
|
||||
|
|
|
|||
|
|
@ -272,8 +272,6 @@ export class LLMRequest extends Schema.Class<LLMRequest>("LLM.Request")({
|
|||
providerOptions: Schema.optional(ProviderOptions),
|
||||
http: Schema.optional(HttpOptions),
|
||||
cache: Schema.optional(CachePolicy),
|
||||
// Stable cache affinity for protocols that support provider-managed prompt caching.
|
||||
promptCacheKey: Schema.optional(Schema.String),
|
||||
metadata: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
}) {}
|
||||
|
||||
|
|
@ -291,7 +289,6 @@ export namespace LLMRequest {
|
|||
providerOptions: request.providerOptions,
|
||||
http: request.http,
|
||||
cache: request.cache,
|
||||
promptCacheKey: request.promptCacheKey,
|
||||
metadata: request.metadata,
|
||||
})
|
||||
|
||||
|
|
|
|||
|
|
@ -256,18 +256,17 @@ export class CacheHint extends Schema.Class<CacheHint>("LLM.CacheHint")({
|
|||
ttlSeconds: Schema.optional(Schema.Number),
|
||||
}) {}
|
||||
|
||||
// Auto-placement policy for prompt caching. The protocol-neutral lowering step
|
||||
// reads this and injects `CacheHint`s at the configured boundaries; the
|
||||
// per-protocol body builders then translate those hints into wire markers as
|
||||
// usual. `"auto"` is the recommended default for agent loops — it places
|
||||
// breakpoints at the last tool definition, the first and last distinct system
|
||||
// parts, and the conversation tail. The rolling message breakpoint keeps a
|
||||
// prior cache entry within Anthropic/Bedrock's 20-block lookback during long
|
||||
// tool loops.
|
||||
//
|
||||
// Pass `"none"` to opt out entirely (the legacy behavior). Pass the granular
|
||||
// object form to override individual choices.
|
||||
export const CachePolicyObject = Schema.Struct({
|
||||
const CacheKey = { key: Schema.optional(Schema.String) }
|
||||
|
||||
export const CacheAuto = Schema.Struct({
|
||||
mode: Schema.Literal("auto"),
|
||||
...CacheKey,
|
||||
})
|
||||
export type CacheAuto = Schema.Schema.Type<typeof CacheAuto>
|
||||
|
||||
export const CacheExplicit = Schema.Struct({
|
||||
mode: Schema.Literal("explicit"),
|
||||
...CacheKey,
|
||||
tools: Schema.optional(Schema.Boolean),
|
||||
system: Schema.optional(Schema.Boolean),
|
||||
messages: Schema.optional(
|
||||
|
|
@ -279,7 +278,12 @@ export const CachePolicyObject = Schema.Struct({
|
|||
),
|
||||
ttlSeconds: Schema.optional(Schema.Number),
|
||||
})
|
||||
export type CachePolicyObject = Schema.Schema.Type<typeof CachePolicyObject>
|
||||
export type CacheExplicit = Schema.Schema.Type<typeof CacheExplicit>
|
||||
|
||||
export const CachePolicy = Schema.Union([Schema.Literal("auto"), Schema.Literal("none"), CachePolicyObject])
|
||||
// Omitted configuration uses automatic provider behavior and OpenCode's
|
||||
// automatic breakpoint placement where required. `"none"` sends no cache key
|
||||
// or explicit controls; providers may still cache implicitly.
|
||||
export const CachePolicy = Schema.Union([Schema.Literal("none"), CacheAuto, CacheExplicit])
|
||||
export type CachePolicy = Schema.Schema.Type<typeof CachePolicy>
|
||||
|
||||
export const cacheKey = (cache: CachePolicy | undefined) => (cache && cache !== "none" ? cache.key : undefined)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue