feat: drive reasoning variants from models.dev reasoning_options
Parse the curated reasoning_options field from models.dev api.json and use its effort values to generate reasoning variants instead of the hardcoded per-package tables, in both the v1 provider catalog and the v2 catalog plugin. - core: ModelsDev.ReasoningOption discriminated union (toggle | effort | budget_tokens); effort values stay open strings and unknown option types are tolerated since api.json is cast, not decoded - core: ReasoningVariants shared per-package effort encoder used by v1 ProviderTransform.variants and the v2 ModelsDevPlugin - core: v2 catalog generates effort variants from reasoning_options; curated experimental modes win id collisions; anthropic profile gains the effort semantic - llm: anthropic protocol supports adaptive thinking, lowers effort to output_config.effort, and sends the effort-2025-11-24 beta header - opencode: resolved Provider.Model carries capabilities.reasoningOptions (unknown types and null effort values dropped at the mapping boundary); config models accept reasoning_options; models without usable effort data fall back to the hardcoded tables unchanged Catalog-wide audit vs live api.json: 2975 models byte-identical, 38 diffs, all data correcting stale hardcoded effort lists.
This commit is contained in:
parent
c939aa04fe
commit
f7dcfc2680
15 changed files with 954 additions and 62 deletions
|
|
@ -57,6 +57,58 @@ describe("Anthropic Messages route", () => {
|
|||
}),
|
||||
)
|
||||
|
||||
it.effect("lowers enabled thinking to a budget", () =>
|
||||
Effect.gen(function* () {
|
||||
const prepared = yield* LLMClient.prepare<AnthropicMessages.AnthropicMessagesBody>(
|
||||
LLM.updateRequest(request, {
|
||||
providerOptions: { anthropic: { thinking: { type: "enabled", budgetTokens: 16_000 } } },
|
||||
}),
|
||||
)
|
||||
expect(prepared.body.thinking).toEqual({ type: "enabled", budget_tokens: 16_000 })
|
||||
expect(prepared.body.output_config).toBeUndefined()
|
||||
}),
|
||||
)
|
||||
|
||||
it.effect("lowers adaptive thinking and effort to output_config", () =>
|
||||
Effect.gen(function* () {
|
||||
const prepared = yield* LLMClient.prepare<AnthropicMessages.AnthropicMessagesBody>(
|
||||
LLM.updateRequest(request, {
|
||||
providerOptions: {
|
||||
anthropic: { thinking: { type: "adaptive", display: "summarized" }, effort: "high" },
|
||||
},
|
||||
}),
|
||||
)
|
||||
expect(prepared.body.thinking).toEqual({ type: "adaptive", display: "summarized" })
|
||||
expect(prepared.body.output_config).toEqual({ effort: "high" })
|
||||
}),
|
||||
)
|
||||
|
||||
it.effect("adds the effort beta header only when effort is set", () =>
|
||||
Effect.gen(function* () {
|
||||
const seen: Record<string, string>[] = []
|
||||
const body = () =>
|
||||
sseEvents(
|
||||
{ type: "message_start", message: { usage: { input_tokens: 1 } } },
|
||||
{ type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 1 } },
|
||||
{ type: "message_stop" },
|
||||
)
|
||||
const layer = dynamicResponse((input) =>
|
||||
Effect.sync(() => {
|
||||
seen.push({ ...input.request.headers })
|
||||
return input.respond(body(), { headers: { "content-type": "text/event-stream" } })
|
||||
}),
|
||||
)
|
||||
yield* LLMClient.generate(
|
||||
LLM.updateRequest(request, { providerOptions: { anthropic: { effort: "high" } } }),
|
||||
).pipe(Effect.provide(layer))
|
||||
yield* LLMClient.generate(request).pipe(Effect.provide(layer))
|
||||
|
||||
expect(seen[0]["anthropic-version"]).toBe("2023-06-01")
|
||||
expect(seen[0]["anthropic-beta"]).toBe("effort-2025-11-24")
|
||||
expect(seen[1]["anthropic-beta"]).toBeUndefined()
|
||||
}),
|
||||
)
|
||||
|
||||
it.effect("lowers chronological system updates natively for Claude Opus 4.8 with cache hints", () =>
|
||||
Effect.gen(function* () {
|
||||
const prepared = yield* LLMClient.prepare<AnthropicMessages.AnthropicMessagesBody>(
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue