fix(tui): flag likely cache busts accurately (#38727)

This commit is contained in:
Kit Langton 2026-07-24 14:35:13 -04:00 committed by GitHub
commit 3193f3aa95
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 131 additions and 20 deletions

View file

@ -81,7 +81,15 @@ import { PluginSlot } from "../../plugin/context"
import { Keymap, type KeymapCommand } from "../../context/keymap"
import { usePathFormatter } from "../../context/path-format"
import { useLocation } from "../../context/location"
import { createSessionRows, messageBoundaryIDs, resolvePart, type PartRef, type SessionRow } from "./rows"
import {
cacheReuseDrop,
createSessionRows,
messageBoundaryIDs,
resolvePart,
type CacheUsage,
type PartRef,
type SessionRow,
} from "./rows"
import { switchLabel } from "../../util/model"
import { findMessageBoundary, messageNavigationSlack } from "./message-navigation"
import { stringWidth } from "../../util/string-width"
@ -1079,7 +1087,7 @@ function SessionRowView(props: SessionRowViewProps) {
{(row) => (
<TurnTokenUsage
messageIDs={row().messageIDs}
previousCacheRead={row().previousCacheRead}
previousCache={row().previousCache}
message={props.message}
/>
)}
@ -1091,13 +1099,13 @@ function SessionRowView(props: SessionRowViewProps) {
function TurnTokenUsage(props: {
messageIDs: string[]
previousCacheRead?: number
previousCache?: CacheUsage
message: (messageID: string) => SessionMessageInfo | undefined
}) {
const config = useConfig()
const { themeV2 } = useTheme()
const steps = createMemo(() => {
let previousCacheRead = props.previousCacheRead
let previousCache = props.previousCache
return props.messageIDs.flatMap((messageID) => {
const message = props.message(messageID)
if (message?.type !== "assistant" || !message.tokens) return []
@ -1109,18 +1117,16 @@ function TurnTokenUsage(props: {
message.tokens.cache.write
if (total === 0) return []
const newTokens = total - message.tokens.cache.read
const cacheBust =
previousCacheRead !== undefined && message.tokens.cache.read < previousCacheRead
? previousCacheRead - message.tokens.cache.read
: undefined
previousCacheRead = message.tokens.cache.read
const currentCache = { read: message.tokens.cache.read, model: message.model }
const reuseDrop = cacheReuseDrop(previousCache, currentCache)
previousCache = currentCache
return [
{
finish: message.finish === "tool-calls" ? "tool-call" : (message.finish ?? "unknown"),
newTokens,
cached: message.tokens.cache.read,
total,
cacheBust,
reuseDrop,
},
]
})
@ -1165,9 +1171,9 @@ function TurnTokenUsage(props: {
{" "}
{item.total.toLocaleString().padStart(columns().total)}
</text>
<Show when={item.cacheBust !== undefined}>
<text fg={themeV2.text.feedback.error.default}>
! Cache bust: {item.cacheBust?.toLocaleString()} fewer cached tokens than the previous step
<Show when={item.reuseDrop !== undefined}>
<text fg={themeV2.text.feedback.warning.default}>
! Likely cache bust: {item.reuseDrop?.toLocaleString()} fewer cached tokens than the previous step
</text>
</Show>
</box>

View file

@ -10,6 +10,11 @@ export type PartRef = {
partID: string
}
export type CacheUsage = {
read: number
model: SessionMessageAssistant["model"]
}
export type SessionRow =
| { type: "message"; messageID: string }
| { type: "compaction-queued"; inputID: string }
@ -28,7 +33,7 @@ export type SessionRow =
completed: boolean
}
| { type: "assistant-footer"; messageID: string }
| { type: "turn-usage"; messageIDs: string[]; previousCacheRead?: number }
| { type: "turn-usage"; messageIDs: string[]; previousCache?: CacheUsage }
export function createSessionRows(sessionID: Accessor<string>) {
const data = useData()
@ -280,7 +285,7 @@ export function reduceSessionRows(
const pendingCompactions = messages.filter((message) => message.type === "compaction" && message.status === "running")
const pending = new Set([...pendingCompactions.map((message) => message.id), ...inputs])
const usage = turnTokens
? { steps: [] as SessionMessageAssistant[], previousTurnCacheRead: undefined as number | undefined }
? { steps: [] as SessionMessageAssistant[], previousTurnCache: undefined as CacheUsage | undefined }
: undefined
return [
...messages.filter((message) => !pending.has(message.id)),
@ -289,6 +294,8 @@ export function reduceSessionRows(
].reduce<SessionRow[]>((rows, message) => {
if (message.type !== "assistant") {
if (message.type === "synthetic" && !message.description?.trim()) return rows
if (message.type === "compaction" && message.status === "completed" && usage)
usage.previousTurnCache = undefined
if (!pending.has(message.id)) completePrevious(rows)
rows.push({ type: "message", messageID: message.id })
return rows
@ -312,11 +319,9 @@ export function reduceSessionRows(
rows.push({
type: "turn-usage",
messageIDs: stepsWithUsage.map((step) => step.id),
...(usage.previousTurnCacheRead === undefined
? {}
: { previousCacheRead: usage.previousTurnCacheRead }),
...(usage.previousTurnCache === undefined ? {} : { previousCache: usage.previousTurnCache }),
})
usage.previousTurnCacheRead = last.tokens.cache.read
usage.previousTurnCache = { read: last.tokens.cache.read, model: last.model }
}
usage.steps.length = 0
}
@ -324,6 +329,20 @@ export function reduceSessionRows(
}, [])
}
export function cacheReuseDrop(previous: CacheUsage | undefined, current: CacheUsage) {
if (previous === undefined) return
if (
previous.model.providerID !== current.model.providerID ||
previous.model.id !== current.model.id ||
previous.model.variant !== current.model.variant
)
return
const drop = previous.read - current.read
// OpenAI cache reads can move between one and two 1,024-token buckets without a material loss of reuse.
if (current.model.providerID === "openai" && drop >= 1_024 && drop <= 2_048) return
return drop > 0 ? drop : undefined
}
function hasTokenUsage(
message: SessionMessageAssistant,
): message is SessionMessageAssistant & { tokens: NonNullable<SessionMessageAssistant["tokens"]> } {

View file

@ -1,6 +1,92 @@
import { expect, test } from "bun:test"
import type { SessionMessageAssistant, SessionMessageInfo } from "@opencode-ai/client"
import { messageBoundaryIDs, reduceSessionRows } from "../../../src/routes/session/rows"
import { cacheReuseDrop, messageBoundaryIDs, reduceSessionRows } from "../../../src/routes/session/rows"
test("filters OpenAI cache quantization from cache reuse drops", () => {
const openai = { id: "gpt", providerID: "openai" }
expect(cacheReuseDrop(undefined, { read: 10_000, model: openai })).toBeUndefined()
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 11_000, model: openai })).toBeUndefined()
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 8_977, model: openai })).toBe(1_023)
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 8_976, model: openai })).toBeUndefined()
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 8_500, model: openai })).toBeUndefined()
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 7_952, model: openai })).toBeUndefined()
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 7_951, model: openai })).toBe(2_049)
})
test("compares cache reuse only for the same model", () => {
const previous = { read: 10_000, model: { id: "claude", providerID: "anthropic" } }
expect(cacheReuseDrop(previous, { read: 8_976, model: { id: "gpt", providerID: "openai" } })).toBeUndefined()
expect(cacheReuseDrop(previous, { read: 8_976, model: { id: "claude", providerID: "anthropic" } })).toBe(1_024)
expect(
cacheReuseDrop(
{ read: 10_000, model: { id: "gpt", providerID: "openai", variant: "low" } },
{ read: 8_976, model: { id: "gpt", providerID: "openai", variant: "high" } },
),
).toBeUndefined()
})
test("carries model identity with the cross-turn cache baseline", () => {
const first = assistant("assistant-1", [])
first.model = { id: "claude", providerID: "anthropic" }
first.finish = "stop"
first.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 10_000, write: 0 } }
const second = assistant("assistant-2", [])
second.model = { id: "gpt", providerID: "openai" }
second.finish = "stop"
second.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 8_976, write: 0 } }
const rows = reduceSessionRows(
[
{ type: "user", id: "user-1", text: "First", time: { created: 0 } },
first,
{ type: "user", id: "user-2", text: "Second", time: { created: 2 } },
second,
],
new Set(),
true,
).filter((row) => row.type === "turn-usage")
expect(rows).toEqual([
{ type: "turn-usage", messageIDs: ["assistant-1"] },
{
type: "turn-usage",
messageIDs: ["assistant-2"],
previousCache: { read: 10_000, model: { id: "claude", providerID: "anthropic" } },
},
])
})
test("resets the cross-turn cache baseline after compaction", () => {
const first = assistant("assistant-1", [])
first.finish = "stop"
first.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 370_176, write: 0 } }
const second = assistant("assistant-2", [])
second.finish = "stop"
second.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 13_824, write: 0 } }
const rows = reduceSessionRows(
[
first,
{
type: "compaction",
id: "compaction-1",
status: "completed",
reason: "auto",
summary: "Compacted context",
recent: "",
time: { created: 2 },
},
second,
],
new Set(),
true,
).filter((row) => row.type === "turn-usage")
expect(rows).toEqual([
{ type: "turn-usage", messageIDs: ["assistant-1"] },
{ type: "turn-usage", messageIDs: ["assistant-2"] },
])
})
test("assigns assistant boundaries to the first rendered row instead of the first text row", () => {
const messages: SessionMessageInfo[] = [