Compare commits

...
Author SHA1 Message Date
Aiden Cline f0303f362a fix(ai): emit cache markers for Alibaba Chat 2026-09-29 21:16:20 -05:00
5 changed files with 37 additions and 13 deletions

No files matched your search

+14 -11
View File
@@ -20,6 +20,7 @@ const AUTO: CachePolicyObject = {
messages: { tail: 1 },
}
const MESSAGE_PREFIX: CachePolicyObject = { system: true, messages: { tail: 1 } }
const NONE: CachePolicyObject = {}
const BREAKPOINT_CAP = 4
@@ -28,8 +29,8 @@ const BREAKPOINT_CAP = 4
// - "auto" → tools + first/last system + final message boundary.
// - "none" → no auto placement; manual `CacheHint`s still flow.
// - object form → exactly what the caller asked for.
const resolve = (policy: CachePolicy | undefined): CachePolicyObject => {
if (policy === undefined || policy === "auto") return AUTO
const resolve = (policy: CachePolicy | undefined, automatic = AUTO): CachePolicyObject => {
if (policy === undefined || policy === "auto") return automatic
if (policy === "none") return NONE
return policy
}
@@ -38,6 +39,7 @@ const resolve = (policy: CachePolicy | undefined): CachePolicyObject => {
// prefix caching, Gemini's implicit + out-of-band CachedContent). Skip the
// whole policy pass for these — emitting hints would be harmless but pointless.
const RESPECTS_INLINE_HINTS = new Set([
"alibaba-chat",
"alibaba-messages",
"anthropic-messages",
"anthropic-compatible-messages",
@@ -58,7 +60,7 @@ const openRouterPolicy = (modelID: string): CachePolicyObject => {
// `~anthropic/claude-sonnet-latest` style IDs are OpenRouter aliases for the latest model in a family.
const id = modelID.replace(/^~/, "")
if (id.startsWith("anthropic/")) return AUTO
if (id.startsWith("qwen/")) return { system: true, messages: { tail: 1 } }
if (id.startsWith("qwen/")) return MESSAGE_PREFIX
return NONE
}
@@ -152,8 +154,8 @@ const markMessages = (
return next
}
const countHints = (request: LLMRequest) =>
countToolHints(request.tools) +
const countHints = (request: LLMRequest, tools = true) =>
(tools ? countToolHints(request.tools) : 0) +
request.system.reduce((count, part) => count + (part.cache === undefined ? 0 : 1), 0) +
request.messages.reduce(
(count, message) =>
@@ -167,15 +169,16 @@ const countHints = (request: LLMRequest) =>
export const applyCachePolicy = (request: LLMRequest): LLMRequest => {
if (!RESPECTS_INLINE_HINTS.has(request.model.route.id)) return request
const policy =
request.model.route.id === "openrouter" && (request.cache === undefined || request.cache === "auto")
? openRouterPolicy(request.model.id)
: resolve(request.cache)
const cacheTools = request.model.route.id !== "alibaba-chat"
const policy = resolve(
request.cache,
request.model.route.id === "openrouter" ? openRouterPolicy(request.model.id) : cacheTools ? AUTO : MESSAGE_PREFIX,
)
if (!policy.tools && !policy.system && !policy.messages) return request
const hint = makeHint(policy.ttlSeconds)
const budget = { remaining: Math.max(0, BREAKPOINT_CAP - countHints(request)) }
const tools = policy.tools ? markLastTool(request.tools, hint, budget) : request.tools
const budget = { remaining: Math.max(0, BREAKPOINT_CAP - countHints(request, cacheTools)) }
const tools = policy.tools && cacheTools ? markLastTool(request.tools, hint, budget) : request.tools
const system = policy.system ? markSystemBoundaries(request.system, hint, budget) : request.system
const messages = policy.messages ? markMessages(request.messages, policy.messages, hint, budget) : request.messages
+11 -1
View File
@@ -4,6 +4,7 @@ import type { LanguageModelCompatibility } from "../schema/index.js"
import { OpenAIChat } from "./openai-chat.js"
import { JsonObject, ProviderShared } from "./shared.js"
import { OpenResponsesOptions } from "./utils/open-responses-options.js"
import { newBreakpoints } from "./utils/cache.js"
export type ReasoningEffort = OpenResponsesOptions.ReasoningEffort
@@ -67,8 +68,17 @@ export const protocol = Protocol.make({
}),
from: Effect.fn("AlibabaChat.fromRequest")(function* (req) {
const opts = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Options))(req.providerOptions ?? {})
const breakpoints = newBreakpoints(4)
return {
...(yield* OpenAIChat.protocol.body.from(req)),
...(yield* OpenAIChat.fromRequest(req, {
// Alibaba caches tool definitions with system content and only supports 5-minute markers.
cacheTools: false,
cacheControl: (cache) => {
if (cache === undefined || breakpoints.remaining === 0) return undefined
breakpoints.remaining -= 1
return { type: "ephemeral" }
},
})),
enable_thinking: opts.enableThinking,
// Alibaba also rejects an explicit budget that is not below `max_completion_tokens`.
thinking_budget:
+2 -1
View File
@@ -330,6 +330,7 @@ export interface ParserState {
// fields into `LLMRequest`.
interface LoweringOptions {
readonly cacheControl?: (cache: CacheHint | undefined) => OpenAIChatCacheControl | undefined
readonly cacheTools?: boolean
readonly toolCallID?: (id: string) => string
}
@@ -341,7 +342,7 @@ const lowerTool = (tool: ToolDefinition, options: LoweringOptions, supportsStric
parameters: tool.inputSchema,
...(supportsStrictMode ? { strict: false } : {}),
},
cache_control: options.cacheControl?.(tool.cache),
cache_control: options.cacheTools === false ? undefined : options.cacheControl?.(tool.cache),
})
const lowerToolChoice = (toolChoice: NonNullable<LLMRequest["toolChoice"]>) =>
@@ -17,6 +17,8 @@ const record = (api: "chat" | "messages" | "responses") =>
})
for (const api of ["chat", "messages", "responses"] as const) {
// Preserve the request shape of Chat recordings made before explicit caching was enabled.
const cache = api === "chat" ? "none" : undefined
const recorded = record(api)
describe(`Alibaba ${api} capabilities`, () => {
for (const enabled of [false, true]) {
@@ -27,6 +29,7 @@ for (const api of ["chat", "messages", "responses"] as const) {
Effect.gen(function* () {
const request = LLM.request({
model: alibaba[api]("qwen3.7-plus"),
cache,
providerOptions:
api === "messages"
? { thinking: { type: enabled ? "enabled" : "disabled", ...(enabled ? { budgetTokens: 1024 } : {}) } }
@@ -61,6 +64,7 @@ for (const api of ["chat", "messages", "responses"] as const) {
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba[api]("qwen3.8-flash"),
cache,
providerOptions: api === "messages" ? { thinking: { type: "disabled" } } : { enableThinking: false },
messages: [
Message.user([
@@ -84,6 +88,7 @@ for (const api of ["chat", "messages", "responses"] as const) {
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba[api]("qwen3.8-max"),
cache,
prompt: "Find the current weather in Paris.",
providerOptions: api === "messages" ? { thinking: { type: "disabled" } } : { reasoningEffort: "none" },
tools: [
@@ -118,6 +123,7 @@ record("chat").effect.with(
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba.chat("qwen3.8-max"),
cache: "none",
prompt: 'Return a JSON object with one key "city" set to the capital city of France.',
providerOptions: { reasoningEffort: "none", responseFormat: { type: "json_object" } },
generation: { maxTokens: 1024 },
@@ -19,6 +19,8 @@ const weather = ToolDefinition.make({
})
for (const api of ["chat", "messages", "responses"] as const) {
// Preserve the request shape of Chat recordings made before explicit caching was enabled.
const cache = api === "chat" ? "none" : undefined
const recorded = recordedTests({
prefix: `alibaba-${api}`,
provider: "alibaba",
@@ -37,6 +39,7 @@ for (const api of ["chat", "messages", "responses"] as const) {
Effect.gen(function* () {
const request = LLM.request({
model: alibaba[api]("qwen3.8-max"),
cache,
prompt: "What is 173 multiplied by 219? Reply with only the final integer.",
providerOptions: api === "messages" ? { effort } : { reasoningEffort: effort },
generation: { maxTokens: 4096 },
@@ -68,6 +71,7 @@ for (const api of ["chat", "messages", "responses"] as const) {
Effect.gen(function* () {
const request = LLM.request({
model: alibaba[api]("qwen3.8-max"),
cache,
providerOptions:
api === "messages"
? { effort: "medium" }