fix(tui): flag likely cache busts accurately (#38727)
This commit is contained in:
parent
ee5460a152
commit
3193f3aa95
3 changed files with 131 additions and 20 deletions
|
|
@ -81,7 +81,15 @@ import { PluginSlot } from "../../plugin/context"
|
|||
import { Keymap, type KeymapCommand } from "../../context/keymap"
|
||||
import { usePathFormatter } from "../../context/path-format"
|
||||
import { useLocation } from "../../context/location"
|
||||
import { createSessionRows, messageBoundaryIDs, resolvePart, type PartRef, type SessionRow } from "./rows"
|
||||
import {
|
||||
cacheReuseDrop,
|
||||
createSessionRows,
|
||||
messageBoundaryIDs,
|
||||
resolvePart,
|
||||
type CacheUsage,
|
||||
type PartRef,
|
||||
type SessionRow,
|
||||
} from "./rows"
|
||||
import { switchLabel } from "../../util/model"
|
||||
import { findMessageBoundary, messageNavigationSlack } from "./message-navigation"
|
||||
import { stringWidth } from "../../util/string-width"
|
||||
|
|
@ -1079,7 +1087,7 @@ function SessionRowView(props: SessionRowViewProps) {
|
|||
{(row) => (
|
||||
<TurnTokenUsage
|
||||
messageIDs={row().messageIDs}
|
||||
previousCacheRead={row().previousCacheRead}
|
||||
previousCache={row().previousCache}
|
||||
message={props.message}
|
||||
/>
|
||||
)}
|
||||
|
|
@ -1091,13 +1099,13 @@ function SessionRowView(props: SessionRowViewProps) {
|
|||
|
||||
function TurnTokenUsage(props: {
|
||||
messageIDs: string[]
|
||||
previousCacheRead?: number
|
||||
previousCache?: CacheUsage
|
||||
message: (messageID: string) => SessionMessageInfo | undefined
|
||||
}) {
|
||||
const config = useConfig()
|
||||
const { themeV2 } = useTheme()
|
||||
const steps = createMemo(() => {
|
||||
let previousCacheRead = props.previousCacheRead
|
||||
let previousCache = props.previousCache
|
||||
return props.messageIDs.flatMap((messageID) => {
|
||||
const message = props.message(messageID)
|
||||
if (message?.type !== "assistant" || !message.tokens) return []
|
||||
|
|
@ -1109,18 +1117,16 @@ function TurnTokenUsage(props: {
|
|||
message.tokens.cache.write
|
||||
if (total === 0) return []
|
||||
const newTokens = total - message.tokens.cache.read
|
||||
const cacheBust =
|
||||
previousCacheRead !== undefined && message.tokens.cache.read < previousCacheRead
|
||||
? previousCacheRead - message.tokens.cache.read
|
||||
: undefined
|
||||
previousCacheRead = message.tokens.cache.read
|
||||
const currentCache = { read: message.tokens.cache.read, model: message.model }
|
||||
const reuseDrop = cacheReuseDrop(previousCache, currentCache)
|
||||
previousCache = currentCache
|
||||
return [
|
||||
{
|
||||
finish: message.finish === "tool-calls" ? "tool-call" : (message.finish ?? "unknown"),
|
||||
newTokens,
|
||||
cached: message.tokens.cache.read,
|
||||
total,
|
||||
cacheBust,
|
||||
reuseDrop,
|
||||
},
|
||||
]
|
||||
})
|
||||
|
|
@ -1165,9 +1171,9 @@ function TurnTokenUsage(props: {
|
|||
{" "}
|
||||
{item.total.toLocaleString().padStart(columns().total)}
|
||||
</text>
|
||||
<Show when={item.cacheBust !== undefined}>
|
||||
<text fg={themeV2.text.feedback.error.default}>
|
||||
! Cache bust: {item.cacheBust?.toLocaleString()} fewer cached tokens than the previous step
|
||||
<Show when={item.reuseDrop !== undefined}>
|
||||
<text fg={themeV2.text.feedback.warning.default}>
|
||||
! Likely cache bust: {item.reuseDrop?.toLocaleString()} fewer cached tokens than the previous step
|
||||
</text>
|
||||
</Show>
|
||||
</box>
|
||||
|
|
|
|||
|
|
@ -10,6 +10,11 @@ export type PartRef = {
|
|||
partID: string
|
||||
}
|
||||
|
||||
export type CacheUsage = {
|
||||
read: number
|
||||
model: SessionMessageAssistant["model"]
|
||||
}
|
||||
|
||||
export type SessionRow =
|
||||
| { type: "message"; messageID: string }
|
||||
| { type: "compaction-queued"; inputID: string }
|
||||
|
|
@ -28,7 +33,7 @@ export type SessionRow =
|
|||
completed: boolean
|
||||
}
|
||||
| { type: "assistant-footer"; messageID: string }
|
||||
| { type: "turn-usage"; messageIDs: string[]; previousCacheRead?: number }
|
||||
| { type: "turn-usage"; messageIDs: string[]; previousCache?: CacheUsage }
|
||||
|
||||
export function createSessionRows(sessionID: Accessor<string>) {
|
||||
const data = useData()
|
||||
|
|
@ -280,7 +285,7 @@ export function reduceSessionRows(
|
|||
const pendingCompactions = messages.filter((message) => message.type === "compaction" && message.status === "running")
|
||||
const pending = new Set([...pendingCompactions.map((message) => message.id), ...inputs])
|
||||
const usage = turnTokens
|
||||
? { steps: [] as SessionMessageAssistant[], previousTurnCacheRead: undefined as number | undefined }
|
||||
? { steps: [] as SessionMessageAssistant[], previousTurnCache: undefined as CacheUsage | undefined }
|
||||
: undefined
|
||||
return [
|
||||
...messages.filter((message) => !pending.has(message.id)),
|
||||
|
|
@ -289,6 +294,8 @@ export function reduceSessionRows(
|
|||
].reduce<SessionRow[]>((rows, message) => {
|
||||
if (message.type !== "assistant") {
|
||||
if (message.type === "synthetic" && !message.description?.trim()) return rows
|
||||
if (message.type === "compaction" && message.status === "completed" && usage)
|
||||
usage.previousTurnCache = undefined
|
||||
if (!pending.has(message.id)) completePrevious(rows)
|
||||
rows.push({ type: "message", messageID: message.id })
|
||||
return rows
|
||||
|
|
@ -312,11 +319,9 @@ export function reduceSessionRows(
|
|||
rows.push({
|
||||
type: "turn-usage",
|
||||
messageIDs: stepsWithUsage.map((step) => step.id),
|
||||
...(usage.previousTurnCacheRead === undefined
|
||||
? {}
|
||||
: { previousCacheRead: usage.previousTurnCacheRead }),
|
||||
...(usage.previousTurnCache === undefined ? {} : { previousCache: usage.previousTurnCache }),
|
||||
})
|
||||
usage.previousTurnCacheRead = last.tokens.cache.read
|
||||
usage.previousTurnCache = { read: last.tokens.cache.read, model: last.model }
|
||||
}
|
||||
usage.steps.length = 0
|
||||
}
|
||||
|
|
@ -324,6 +329,20 @@ export function reduceSessionRows(
|
|||
}, [])
|
||||
}
|
||||
|
||||
export function cacheReuseDrop(previous: CacheUsage | undefined, current: CacheUsage) {
|
||||
if (previous === undefined) return
|
||||
if (
|
||||
previous.model.providerID !== current.model.providerID ||
|
||||
previous.model.id !== current.model.id ||
|
||||
previous.model.variant !== current.model.variant
|
||||
)
|
||||
return
|
||||
const drop = previous.read - current.read
|
||||
// OpenAI cache reads can move between one and two 1,024-token buckets without a material loss of reuse.
|
||||
if (current.model.providerID === "openai" && drop >= 1_024 && drop <= 2_048) return
|
||||
return drop > 0 ? drop : undefined
|
||||
}
|
||||
|
||||
function hasTokenUsage(
|
||||
message: SessionMessageAssistant,
|
||||
): message is SessionMessageAssistant & { tokens: NonNullable<SessionMessageAssistant["tokens"]> } {
|
||||
|
|
|
|||
|
|
@ -1,6 +1,92 @@
|
|||
import { expect, test } from "bun:test"
|
||||
import type { SessionMessageAssistant, SessionMessageInfo } from "@opencode-ai/client"
|
||||
import { messageBoundaryIDs, reduceSessionRows } from "../../../src/routes/session/rows"
|
||||
import { cacheReuseDrop, messageBoundaryIDs, reduceSessionRows } from "../../../src/routes/session/rows"
|
||||
|
||||
test("filters OpenAI cache quantization from cache reuse drops", () => {
|
||||
const openai = { id: "gpt", providerID: "openai" }
|
||||
expect(cacheReuseDrop(undefined, { read: 10_000, model: openai })).toBeUndefined()
|
||||
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 11_000, model: openai })).toBeUndefined()
|
||||
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 8_977, model: openai })).toBe(1_023)
|
||||
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 8_976, model: openai })).toBeUndefined()
|
||||
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 8_500, model: openai })).toBeUndefined()
|
||||
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 7_952, model: openai })).toBeUndefined()
|
||||
expect(cacheReuseDrop({ read: 10_000, model: openai }, { read: 7_951, model: openai })).toBe(2_049)
|
||||
})
|
||||
|
||||
test("compares cache reuse only for the same model", () => {
|
||||
const previous = { read: 10_000, model: { id: "claude", providerID: "anthropic" } }
|
||||
expect(cacheReuseDrop(previous, { read: 8_976, model: { id: "gpt", providerID: "openai" } })).toBeUndefined()
|
||||
expect(cacheReuseDrop(previous, { read: 8_976, model: { id: "claude", providerID: "anthropic" } })).toBe(1_024)
|
||||
expect(
|
||||
cacheReuseDrop(
|
||||
{ read: 10_000, model: { id: "gpt", providerID: "openai", variant: "low" } },
|
||||
{ read: 8_976, model: { id: "gpt", providerID: "openai", variant: "high" } },
|
||||
),
|
||||
).toBeUndefined()
|
||||
})
|
||||
|
||||
test("carries model identity with the cross-turn cache baseline", () => {
|
||||
const first = assistant("assistant-1", [])
|
||||
first.model = { id: "claude", providerID: "anthropic" }
|
||||
first.finish = "stop"
|
||||
first.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 10_000, write: 0 } }
|
||||
const second = assistant("assistant-2", [])
|
||||
second.model = { id: "gpt", providerID: "openai" }
|
||||
second.finish = "stop"
|
||||
second.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 8_976, write: 0 } }
|
||||
|
||||
const rows = reduceSessionRows(
|
||||
[
|
||||
{ type: "user", id: "user-1", text: "First", time: { created: 0 } },
|
||||
first,
|
||||
{ type: "user", id: "user-2", text: "Second", time: { created: 2 } },
|
||||
second,
|
||||
],
|
||||
new Set(),
|
||||
true,
|
||||
).filter((row) => row.type === "turn-usage")
|
||||
|
||||
expect(rows).toEqual([
|
||||
{ type: "turn-usage", messageIDs: ["assistant-1"] },
|
||||
{
|
||||
type: "turn-usage",
|
||||
messageIDs: ["assistant-2"],
|
||||
previousCache: { read: 10_000, model: { id: "claude", providerID: "anthropic" } },
|
||||
},
|
||||
])
|
||||
})
|
||||
|
||||
test("resets the cross-turn cache baseline after compaction", () => {
|
||||
const first = assistant("assistant-1", [])
|
||||
first.finish = "stop"
|
||||
first.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 370_176, write: 0 } }
|
||||
const second = assistant("assistant-2", [])
|
||||
second.finish = "stop"
|
||||
second.tokens = { input: 1, output: 0, reasoning: 0, cache: { read: 13_824, write: 0 } }
|
||||
|
||||
const rows = reduceSessionRows(
|
||||
[
|
||||
first,
|
||||
{
|
||||
type: "compaction",
|
||||
id: "compaction-1",
|
||||
status: "completed",
|
||||
reason: "auto",
|
||||
summary: "Compacted context",
|
||||
recent: "",
|
||||
time: { created: 2 },
|
||||
},
|
||||
second,
|
||||
],
|
||||
new Set(),
|
||||
true,
|
||||
).filter((row) => row.type === "turn-usage")
|
||||
|
||||
expect(rows).toEqual([
|
||||
{ type: "turn-usage", messageIDs: ["assistant-1"] },
|
||||
{ type: "turn-usage", messageIDs: ["assistant-2"] },
|
||||
])
|
||||
})
|
||||
|
||||
test("assigns assistant boundaries to the first rendered row instead of the first text row", () => {
|
||||
const messages: SessionMessageInfo[] = [
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue