Compare commits

...
Author SHA1 Message Date
rekram1-node 09b9e4f1bf fix(ai): choose OpenRouter auto caching by model prefix 2026-09-29 17:40:42 +00:00
4 changed files with 83 additions and 5 deletions

No files matched your search

+8 -5
View File
@@ -1,6 +1,8 @@
// Apply an `LLMRequest.cache` policy by injecting `CacheHint`s onto the parts
// the policy designates. Runs once at compile time, before the per-protocol
// body builder, so the existing inline-hint lowering path handles the rest.
// Routes may override auto placement; explicit policies and hints always use
// the caller's placement.
//
// The default `"auto"` shape places breakpoints at the last tool definition,
// the first and last distinct system parts, and the conversation tail. This
@@ -24,8 +26,8 @@ const NONE: CachePolicyObject = {}
const BREAKPOINT_CAP = 4
// Resolution rules:
// - undefined → "auto" — caching is on by default.
// - "auto" → tools + first/last system + final message boundary.
// - undefined → "auto" — use route-appropriate placement by default.
// - "auto" → tools + system + tail, unless the route chooses otherwise.
// - "none" → no auto placement; manual `CacheHint`s still flow.
// - object form → exactly what the caller asked for.
const resolve = (policy: CachePolicy | undefined): CachePolicyObject => {
@@ -156,9 +158,10 @@ const countHints = (request: LLMRequest) =>
export const applyCachePolicy = (request: LLMRequest): LLMRequest => {
if (!RESPECTS_INLINE_HINTS.has(request.model.route.id)) return request
if (request.model.route.id === "openrouter" && (request.cache === undefined || request.cache === "auto"))
return request
const policy = resolve(request.cache)
const policy =
request.model.route.autoCachePolicy && (request.cache === undefined || request.cache === "auto")
? request.model.route.autoCachePolicy(request.model.id)
: resolve(request.cache)
if (!policy.tools && !policy.system && !policy.messages) return request
const hint = makeHint(policy.ttlSeconds)
+12
View File
@@ -4,6 +4,7 @@ import { Endpoint } from "../route/endpoint.js"
import { Protocol } from "../route/protocol.js"
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
import { HttpOptions, ProviderID, type CacheHint, type ModelID, type OpenString } from "../schema/index.js"
import type { CachePolicyObject } from "../schema/options.js"
import type { ProviderPackage } from "../provider-package.js"
import { SystemOne } from "../experimental/system-one.js"
import { OpenAIChat } from "../protocols/openai-chat.js"
@@ -14,6 +15,16 @@ export const id = ProviderID.make("openrouter")
const baseURL = "https://openrouter.ai/api/v1"
const ADAPTER = "openrouter"
const CLAUDE_AUTO: CachePolicyObject = { tools: true, system: true, messages: { tail: 1 } }
const QWEN_AUTO: CachePolicyObject = { system: true, messages: { tail: 1 } }
const autoCachePolicy = (modelID: string): CachePolicyObject => {
if (modelID.startsWith("anthropic/")) return CLAUDE_AUTO
// Alibaba ignores tool-definition markers; content breakpoints work across Qwen models that support them.
if (modelID.startsWith("qwen/")) return QWEN_AUTO
return {}
}
export interface OpenRouterProviderRouting {
readonly [key: string]: unknown
readonly order?: ReadonlyArray<string>
@@ -177,6 +188,7 @@ export const route = Route.make({
id: ADAPTER,
provider: id,
providerMetadataKey: "openrouter",
autoCachePolicy,
protocol,
endpoint: Endpoint.path("/chat/completions", { baseURL }),
framing: OpenAIChat.framing,
+7
View File
@@ -13,6 +13,7 @@ import { sanitizeSurrogates } from "../utils/sanitize.js"
import * as ProviderShared from "../protocols/shared.js"
import { ToolSchemaProjection } from "../protocols/utils/tool-schema.js"
import type { LanguageModelSanitizerCompatibility, ProtocolID, ProviderOptions } from "../schema/index.js"
import type { CachePolicyObject } from "../schema/options.js"
import {
AIError,
CompactionResponse,
@@ -57,6 +58,8 @@ export interface Route<
readonly transport: Transport<Body, Prepared, unknown>
readonly defaults: RouteDefaults
readonly body: RouteBody<Body>
/** Route-owned default cache placement; explicit request policies bypass it. */
readonly autoCachePolicy?: (modelID: string) => CachePolicyObject
readonly supportsEffortUpdates?: (request: LLMRequest) => boolean
readonly sanitizer?: LanguageModelSanitizerCompatibility
readonly with: {
@@ -299,6 +302,7 @@ export interface MakeInput<Body, Frame, Event, State> {
readonly headers?: (input: { readonly request: LLMRequest }) => Record<string, string>
/** Route/request defaults used when compiling requests for this route. */
readonly defaults?: RouteDefaultsInput
readonly autoCachePolicy?: (modelID: string) => CachePolicyObject
}
export interface MakeTransportInput<Body, Prepared, Frame, Event, State> {
@@ -321,6 +325,7 @@ export interface MakeTransportInput<Body, Prepared, Frame, Event, State> {
readonly transport: Transport<Body, Prepared, Frame>
/** Route/request defaults used when compiling requests for this route. */
readonly defaults?: RouteDefaultsInput
readonly autoCachePolicy?: (modelID: string) => CachePolicyObject
}
const streamError = (route: string, message: string, cause: Cause.Cause<unknown>) => {
@@ -389,6 +394,7 @@ function makeFromTransport<Body, Prepared, Frame, Event, State>(
transport: routeInput.transport,
defaults: routeInput.defaults ?? {},
body: protocol.body,
autoCachePolicy: routeInput.autoCachePolicy,
supportsEffortUpdates: protocol.supportsEffortUpdates,
sanitizer: protocol.sanitizer,
with: (patch: RoutePatch<Body, Prepared>) => {
@@ -550,6 +556,7 @@ export function make<Body, Prepared, Frame, Event, State>(
headers: input.headers,
transport: HttpTransport.httpJson({ framing: input.framing }),
defaults: input.defaults,
autoCachePolicy: input.autoCachePolicy,
})
}
@@ -32,6 +32,62 @@ describe("OpenRouter", () => {
}),
)
for (const [modelID, markTools, markContent] of [
["anthropic/claude-sonnet-4.6", true, true],
["qwen/qwen-plus", false, true],
["google/gemini-2.5-flash", false, false],
["openai/gpt-5.6-luna", false, false],
["deepseek/deepseek-v4.1-flash", false, false],
["qwen/qwen3.5-plus-02-15", false, true],
["other/unknown-model", false, false],
] as const) {
it.effect(`selects OpenRouter auto cache placement for ${modelID}`, () =>
Effect.gen(function* () {
const prepared = yield* compileRequest(
LLM.request({
model: OpenRouter.configure({ apiKey: "test-key" }).model(modelID),
system: "Stable prefix",
tools: [{ name: "lookup", description: "Lookup", inputSchema: { type: "object", properties: {} } }],
prompt: "Current turn",
promptCacheKey: "session_123",
}),
)
expect(prepared.body.prompt_cache_key).toBe("session_123")
expect(prepared.body.tools?.[0]?.cache_control).toEqual(markTools ? { type: "ephemeral" } : undefined)
expect(prepared.body.messages).toMatchObject(
markContent
? [
{ role: "system", content: [{ text: "Stable prefix", cache_control: { type: "ephemeral" } }] },
{ role: "user", content: [{ text: "Current turn", cache_control: { type: "ephemeral" } }] },
]
: [
{ role: "system", content: "Stable prefix" },
{ role: "user", content: "Current turn" },
],
)
}),
)
}
it.effect("explicit policy and manual hints bypass the OpenRouter model default", () =>
Effect.gen(function* () {
const prepared = yield* compileRequest(
LLM.request({
model: OpenRouter.configure({ apiKey: "test-key" }).model("google/gemini-2.5-flash"),
system: [{ type: "text", text: "Pinned prefix", cache: new CacheHint({ type: "ephemeral" }) }],
tools: [{ name: "lookup", description: "Lookup", inputSchema: { type: "object", properties: {} } }],
prompt: "Hello",
cache: { tools: true, messages: { tail: 1 } },
}),
)
expect(prepared.body.tools?.[0]?.cache_control).toEqual({ type: "ephemeral" })
expect(prepared.body.messages).toMatchObject([
{ role: "system", content: [{ text: "Pinned prefix", cache_control: { type: "ephemeral" } }] },
{ role: "user", content: [{ text: "Hello", cache_control: { type: "ephemeral" } }] },
])
}),
)
it.effect("lowers the native cache policy to OpenRouter cache controls", () =>
Effect.gen(function* () {
const prepared = yield* compileRequest(