fix(ai): layer prompt cache breakpoints (#38725)
This commit is contained in:
parent
9b640cf97d
commit
993f046dd9
8 changed files with 258 additions and 43 deletions
|
|
@ -39,8 +39,8 @@ describe("applyCachePolicy", () => {
|
|||
}),
|
||||
)
|
||||
|
||||
// No explicit cache field → auto policy fires → last system part + latest
|
||||
// user message both get cache_control markers.
|
||||
// A single system block is both the first and last boundary, so the auto
|
||||
// policy deduplicates it and still marks the conversation tail.
|
||||
expect(prepared.body).toMatchObject({
|
||||
system: [{ type: "text", text: "You are concise.", cache_control: { type: "ephemeral" } }],
|
||||
messages: [{ role: "user", content: [{ type: "text", text: "hi", cache_control: { type: "ephemeral" } }] }],
|
||||
|
|
@ -48,12 +48,15 @@ describe("applyCachePolicy", () => {
|
|||
}),
|
||||
)
|
||||
|
||||
it.effect("'auto' marks the last tool, last system part, and latest user message on Anthropic", () =>
|
||||
it.effect("'auto' marks the last tool, first and last system parts, and final message boundary on Anthropic", () =>
|
||||
Effect.gen(function* () {
|
||||
const prepared = yield* LLMClient.prepare(
|
||||
LLM.request({
|
||||
model: anthropicModel,
|
||||
system: "Sys A",
|
||||
system: [
|
||||
{ type: "text", text: "Base agent" },
|
||||
{ type: "text", text: "Project instructions" },
|
||||
],
|
||||
tools: [{ name: "t1", description: "t1", inputSchema: { type: "object", properties: {} } }],
|
||||
messages: [
|
||||
Message.user("first user"),
|
||||
|
|
@ -66,7 +69,10 @@ describe("applyCachePolicy", () => {
|
|||
|
||||
expect(prepared.body).toMatchObject({
|
||||
tools: [{ name: "t1", cache_control: { type: "ephemeral" } }],
|
||||
system: [{ type: "text", text: "Sys A", cache_control: { type: "ephemeral" } }],
|
||||
system: [
|
||||
{ type: "text", text: "Base agent", cache_control: { type: "ephemeral" } },
|
||||
{ type: "text", text: "Project instructions", cache_control: { type: "ephemeral" } },
|
||||
],
|
||||
messages: [
|
||||
{ role: "user", content: [{ type: "text", text: "first user" }] },
|
||||
{ role: "assistant", content: [{ type: "text", text: "assistant reply" }] },
|
||||
|
|
@ -120,7 +126,10 @@ describe("applyCachePolicy", () => {
|
|||
const prepared = yield* LLMClient.prepare(
|
||||
LLM.request({
|
||||
model: bedrockModel,
|
||||
system: "Sys",
|
||||
system: [
|
||||
{ type: "text", text: "Base agent" },
|
||||
{ type: "text", text: "Project instructions" },
|
||||
],
|
||||
tools: [{ name: "t1", description: "t1", inputSchema: { type: "object", properties: {} } }],
|
||||
messages: [Message.user("first user"), Message.assistant("reply"), Message.user("latest user")],
|
||||
cache: "auto",
|
||||
|
|
@ -131,7 +140,12 @@ describe("applyCachePolicy", () => {
|
|||
toolConfig: {
|
||||
tools: [{ toolSpec: { name: "t1" } }, { cachePoint: { type: "default" } }],
|
||||
},
|
||||
system: [{ text: "Sys" }, { cachePoint: { type: "default" } }],
|
||||
system: [
|
||||
{ text: "Base agent" },
|
||||
{ cachePoint: { type: "default" } },
|
||||
{ text: "Project instructions" },
|
||||
{ cachePoint: { type: "default" } },
|
||||
],
|
||||
messages: [
|
||||
{ role: "user", content: [{ text: "first user" }] },
|
||||
{ role: "assistant", content: [{ text: "reply" }] },
|
||||
|
|
@ -193,9 +207,55 @@ describe("applyCachePolicy", () => {
|
|||
}),
|
||||
)
|
||||
|
||||
const body = prepared.body as { system: Array<{ text: string; cache_control?: unknown }> }
|
||||
const body = prepared.body as {
|
||||
system: Array<{ text: string; cache_control?: unknown }>
|
||||
messages: Array<{ content: Array<{ cache_control?: unknown }> }>
|
||||
}
|
||||
expect(body.system[0]?.cache_control).toEqual({ type: "ephemeral", ttl: "1h" })
|
||||
expect(body.system[1]?.cache_control).toEqual({ type: "ephemeral" })
|
||||
expect(body.messages[0]?.content[0]?.cache_control).toEqual({ type: "ephemeral" })
|
||||
}),
|
||||
)
|
||||
|
||||
it.effect("auto policy stays within the four-breakpoint cap when preserving manual hints", () =>
|
||||
Effect.gen(function* () {
|
||||
const request = LLM.request({
|
||||
model: anthropicModel,
|
||||
system: [
|
||||
{ type: "text", text: "Base agent" },
|
||||
{
|
||||
type: "text",
|
||||
text: "Manual context",
|
||||
cache: new CacheHint({ type: "ephemeral", ttlSeconds: 3600 }),
|
||||
},
|
||||
{ type: "text", text: "Project instructions" },
|
||||
],
|
||||
tools: [{ name: "t1", description: "t1", inputSchema: { type: "object", properties: {} } }],
|
||||
prompt: "hi",
|
||||
cache: "auto",
|
||||
})
|
||||
const applied = applyCachePolicy(request)
|
||||
expect(applied.tools[0]?.cache).toBeDefined()
|
||||
expect(applied.system.map((part) => part.cache !== undefined)).toEqual([true, true, true])
|
||||
const tail = applied.messages[0]!.content[0]!
|
||||
expect("cache" in tail ? tail.cache : undefined).toBeUndefined()
|
||||
expect(applyCachePolicy(applied)).toBe(applied)
|
||||
|
||||
const prepared = yield* LLMClient.prepare(request)
|
||||
|
||||
const body = prepared.body as {
|
||||
tools: Array<{ cache_control?: unknown }>
|
||||
system: Array<{ cache_control?: unknown }>
|
||||
messages: Array<{ content: Array<{ cache_control?: unknown }> }>
|
||||
}
|
||||
const marked = [
|
||||
...body.tools.map((tool) => tool.cache_control),
|
||||
...body.system.map((part) => part.cache_control),
|
||||
...body.messages.flatMap((message) => message.content.map((part) => part.cache_control)),
|
||||
].filter((cache) => cache !== undefined)
|
||||
expect(marked).toHaveLength(4)
|
||||
expect(body.system[1]?.cache_control).toEqual({ type: "ephemeral", ttl: "1h" })
|
||||
expect(body.messages[0]?.content[0]?.cache_control).toBeUndefined()
|
||||
}),
|
||||
)
|
||||
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
|
|
@ -1,6 +1,6 @@
|
|||
import { describe, expect } from "bun:test"
|
||||
import { Effect } from "effect"
|
||||
import { CacheHint, LLM } from "../../src"
|
||||
import { CacheHint, LLM, LLMRequest, Message, ToolCallPart, ToolDefinition } from "../../src"
|
||||
import { LLMClient } from "../../src/route"
|
||||
import * as Anthropic from "../../src/providers/anthropic"
|
||||
import { LARGE_CACHEABLE_SYSTEM } from "../recorded-scenarios"
|
||||
|
|
@ -24,6 +24,39 @@ const cacheRequest = LLM.request({
|
|||
generation: { maxTokens: 16, temperature: 0 },
|
||||
})
|
||||
|
||||
const lookup = ToolDefinition.make({
|
||||
name: "lookup",
|
||||
description: "Look up a fixture value.",
|
||||
inputSchema: {
|
||||
type: "object",
|
||||
properties: { index: { type: "number" } },
|
||||
required: ["index"],
|
||||
additionalProperties: false,
|
||||
},
|
||||
})
|
||||
const longToolTurn = [
|
||||
Message.user("Run the fixture lookups."),
|
||||
...Array.from({ length: 11 }, (_, index) => {
|
||||
const id = `lookup_${index}`
|
||||
return [
|
||||
Message.assistant(ToolCallPart.make({ id, name: lookup.name, input: { index } })),
|
||||
Message.tool({
|
||||
id,
|
||||
name: lookup.name,
|
||||
result: `Fixture result ${index}. `.repeat(80),
|
||||
}),
|
||||
]
|
||||
}).flat(),
|
||||
]
|
||||
const longToolTurnRequest = LLM.request({
|
||||
id: "recorded_anthropic_cache_long_tool_turn",
|
||||
model,
|
||||
system: LARGE_CACHEABLE_SYSTEM,
|
||||
messages: longToolTurn,
|
||||
tools: [lookup],
|
||||
generation: { maxTokens: 16, temperature: 0 },
|
||||
})
|
||||
|
||||
const recorded = recordedTests({
|
||||
prefix: "anthropic-messages-cache",
|
||||
provider: "anthropic",
|
||||
|
|
@ -50,4 +83,28 @@ describe("Anthropic Messages cache recorded", () => {
|
|||
expect(second.usage?.cacheReadInputTokens ?? 0).toBeGreaterThan(0)
|
||||
}),
|
||||
)
|
||||
|
||||
recorded.effect.with("keeps a long tool turn inside the cache lookback", { tags: ["cache", "tool"] }, () =>
|
||||
Effect.gen(function* () {
|
||||
const first = yield* LLMClient.generate(longToolTurnRequest)
|
||||
const firstRead = first.usage?.cacheReadInputTokens ?? 0
|
||||
const firstWrite = first.usage?.cacheWriteInputTokens ?? 0
|
||||
const firstCached = firstRead + firstWrite
|
||||
// The prefix may already be warm when recording, so either a read or a
|
||||
// write establishes that Anthropic recognized the cache boundary.
|
||||
expect(firstCached).toBeGreaterThan(0)
|
||||
|
||||
const second = yield* LLMClient.generate(
|
||||
LLMRequest.update(longToolTurnRequest, {
|
||||
messages: [
|
||||
...longToolTurn,
|
||||
Message.assistant("The fixture lookups are complete."),
|
||||
Message.user("Reply exactly: OK"),
|
||||
],
|
||||
}),
|
||||
)
|
||||
expect(second.usage?.cacheReadInputTokens ?? 0).toBeGreaterThanOrEqual(firstCached)
|
||||
expect(second.usage?.cacheWriteInputTokens ?? 0).toBeLessThan(firstCached)
|
||||
}),
|
||||
)
|
||||
})
|
||||
|
|
|
|||
|
|
@ -939,6 +939,7 @@ describe("Bedrock Converse route", () => {
|
|||
const prepared = yield* LLMClient.prepare<BedrockConverse.BedrockConverseBody>(
|
||||
LLM.request({
|
||||
model,
|
||||
cache: "none",
|
||||
messages: [
|
||||
Message.assistant([ToolCallPart.make({ id: "call_1", name: "read", input: { path: "report.pdf" } })]),
|
||||
Message.tool({
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue