Compare commits

..
616 changed files with 7990 additions and 25113 deletions
-72
View File
@@ -29,78 +29,6 @@ await Effect.runPromise(program.pipe(Effect.provide(llmLayer)))
Run `LLMClient.stream(request)` instead of `generate` when you want incremental `LLMEvent`s. The event stream is provider-neutral — same shape across OpenAI Chat, OpenAI Responses, Anthropic Messages, Gemini, Bedrock Converse, and any OpenAI-compatible deployment.
## Alibaba Cloud Model Studio
`Alibaba` provides standard Model Studio inference. Configure a region explicitly, then select
Chat Completions (`.model` or `.chat`), Anthropic-compatible Messages (`.messages`), or OpenAI-compatible
Responses (`.responses`). These routes use HTTP/SSE.
```ts
import { LLM } from "@opencode/ai"
import { Alibaba } from "@opencode/ai/providers"
const alibaba = Alibaba.configure({
region: "ap-southeast-1", // Singapore
apiKey: process.env.DASHSCOPE_API_KEY,
// workspaceID: "llm-your-workspace", // use a workspace-dedicated endpoint
})
const request = LLM.request({
model: alibaba.model("qwen3.8-max"),
prompt: "Explain this design.",
providerOptions: { reasoningEffort: "medium" },
})
```
### Regions and credentials
| Region | `region` | Shared host when `workspaceID` is omitted |
| ------------------- | ---------------- | ----------------------------------------- |
| Singapore | `ap-southeast-1` | `dashscope-intl.aliyuncs.com` |
| China (Beijing) | `cn-beijing` | `dashscope.aliyuncs.com` |
| China (Hong Kong) | `cn-hongkong` | `cn-hongkong.dashscope.aliyuncs.com` |
| US (Virginia) | `us-east-1` | `dashscope-us.aliyuncs.com` |
| Germany (Frankfurt) | `eu-central-1` | Supply `workspaceID` or `baseURL` |
| Japan (Tokyo) | `ap-northeast-1` | Supply `workspaceID` or `baseURL` |
With `workspaceID`, the host is `{workspaceID}.{region}.maas.aliyuncs.com`. A complete `baseURL`
overrides regional setup, including the API prefix: `/compatible-mode/v1` for Chat/Responses,
or `/apps/anthropic/v1` for Messages. The selector appends its operation path.
Keys and model availability are region-specific. Auth resolves from explicit `auth` or `apiKey`,
then `DASHSCOPE_API_KEY`, then `ALIBABA_API_KEY`.
The access region and inference scope differ: Virginia's `-us` model IDs request US-only inference;
some regions select scope through their workspace. Model IDs pass through unchanged.
Alibaba's [regional guide](https://www.alibabacloud.com/help/en/model-studio/regions) and
[base URL table](https://www.alibabacloud.com/help/en/model-studio/base-url) disagree about Virginia's
shared host; the entry above follows the base URL table. Dedicated hosts can be copied from the console.
### Native options
- **Chat:** `reasoningEffort``reasoning_effort`, `enableThinking``enable_thinking`,
`thinkingBudget``thinking_budget`, and `preserveThinking``preserve_thinking`.
Replay complete `response.message` values to retain `reasoning_content` separately from answer text.
Qwen 3.8 defaults to preserving thinking; older models have different defaults.
Additional options include `toolStream`, `parallelToolCalls`, `repetitionPenalty`, `responseFormat`,
`enableSearch`, and native `searchOptions`. `generation.topK` lowers to `top_k`.
`clearThinking` is a hosted GLM control, and `thinking.type` is available for hosted MiniMax models.
- **Messages:** `effort``output_config.effort`. `thinking.type` accepts enabled/disabled with an
optional `budgetTokens` (or native `budget_tokens`). `outputConfig.format` accepts a JSON schema.
Model Studio's empty thinking signatures are accepted; supplied signatures are replayed unchanged.
- **Responses:** `reasoningEffort``reasoning.effort`, plus `enableThinking`, `store`,
`previousResponseId`, and `conversation`. Omitted `store` retains the API's default (`true`);
set it to `false` for client-managed history. `previousResponseId` requires a stored response.
Hosted tools are `Alibaba.webSearch()`, `Alibaba.webExtractor()`, and `Alibaba.codeInterpreter()`.
Web extraction is used together with web search. Hosted calls/results carry `providerExecuted: true`.
Omitted options preserve provider defaults. Effort values pass through unchanged and accept future
strings. Qwen 3.8 Chat rejects requests combining a thinking budget with effort.
Package entrypoints are `@opencode/ai/providers/alibaba`, `alibaba/chat`, `alibaba/messages`,
and `alibaba/responses`. Live recordings cover all three APIs in Singapore; regional URL construction
is unit-tested for all six regions.
## Z.AI
`ZAI` uses the standard API. Chat Completions is the default language-model API;
-92
View File
@@ -1,92 +0,0 @@
import { Effect, Schema } from "effect"
import { Protocol } from "../route/protocol.js"
import type { LanguageModelCompatibility } from "../schema/index.js"
import { OpenAIChat } from "./openai-chat.js"
import { JsonObject, ProviderShared } from "./shared.js"
import { OpenResponsesOptions } from "./utils/open-responses-options.js"
export type ReasoningEffort = OpenResponsesOptions.ReasoningEffort
const Options = Schema.Struct({
reasoningEffort: OpenResponsesOptions.Options.fields.reasoningEffort,
enableThinking: Schema.optional(Schema.Boolean),
thinkingBudget: Schema.optional(Schema.Int),
preserveThinking: Schema.optional(Schema.Boolean),
clearThinking: Schema.optional(Schema.Boolean),
thinking: Schema.optional(
Schema.Struct({
type: Schema.declare<"adaptive" | "disabled" | (string & {})>(Schema.is(Schema.String)),
}),
),
toolStream: Schema.optional(Schema.Boolean),
parallelToolCalls: OpenResponsesOptions.Options.fields.parallelToolCalls,
repetitionPenalty: Schema.optional(Schema.Number),
responseFormat: Schema.optional(
Schema.Struct({
type: Schema.declare<"text" | "json_object" | "json_schema" | (string & {})>(Schema.is(Schema.String)),
json_schema: Schema.optional(JsonObject),
}),
),
enableSearch: Schema.optional(Schema.Boolean),
searchOptions: Schema.optional(
Schema.Struct({
forced_search: Schema.optional(Schema.Boolean),
search_strategy: Schema.optional(
Schema.declare<"turbo" | "max" | "agent" | "agent_max" | (string & {})>(Schema.is(Schema.String)),
),
enable_search_extension: Schema.optional(Schema.Boolean),
}),
),
})
export type OptionsInput = typeof Options.Type
export const compatibility = {
maxTokensField: "max_completion_tokens",
supportsStore: false,
supportsStrictMode: false,
reasoningField: "reasoning_content",
zaiToolStream: false,
} satisfies LanguageModelCompatibility
export const protocol = Protocol.make({
id: "alibaba-chat",
body: {
schema: Schema.Struct({
...OpenAIChat.bodyFields,
enable_thinking: Options.fields.enableThinking,
thinking_budget: Options.fields.thinkingBudget,
preserve_thinking: Options.fields.preserveThinking,
clear_thinking: Options.fields.clearThinking,
thinking: Options.fields.thinking,
parallel_tool_calls: Options.fields.parallelToolCalls,
repetition_penalty: Options.fields.repetitionPenalty,
top_k: Schema.optional(Schema.Int),
response_format: Options.fields.responseFormat,
enable_search: Options.fields.enableSearch,
search_options: Options.fields.searchOptions,
}),
from: Effect.fn("AlibabaChat.fromRequest")(function* (req) {
const opts = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Options))(req.providerOptions ?? {})
return {
...(yield* OpenAIChat.protocol.body.from(req)),
enable_thinking: opts.enableThinking,
thinking_budget: opts.thinkingBudget,
preserve_thinking: opts.preserveThinking,
clear_thinking: opts.clearThinking,
thinking: opts.thinking,
tool_stream: opts.toolStream,
parallel_tool_calls:
opts.parallelToolCalls ??
(req.toolChoice?.disableParallelToolUse === undefined ? undefined : !req.toolChoice.disableParallelToolUse),
repetition_penalty: opts.repetitionPenalty,
top_k: req.generation?.topK,
response_format: opts.responseFormat,
enable_search: opts.enableSearch,
search_options: opts.searchOptions,
}
}),
},
stream: OpenAIChat.protocol.stream,
})
export * as AlibabaChat from "./alibaba-chat.js"
@@ -1,48 +0,0 @@
import { Effect, Schema } from "effect"
import { Protocol } from "../route/protocol.js"
import { LLMRequest } from "../schema/index.js"
import { AnthropicMessages } from "./anthropic-messages.js"
import { ProviderShared } from "./shared.js"
import { OpenResponsesOptions } from "./utils/open-responses-options.js"
const Options = Schema.Struct({
effort: Schema.optional(OpenResponsesOptions.ReasoningEffort),
thinking: Schema.optional(
Schema.Struct({
type: Schema.declare<"enabled" | "disabled" | (string & {})>(Schema.is(Schema.String)),
budgetTokens: Schema.optional(Schema.Int),
budget_tokens: Schema.optional(Schema.Int),
}),
),
})
export type OptionsInput = typeof Options.Type & Pick<AnthropicMessages.OptionsInput, "outputConfig">
export const protocol = Protocol.make({
id: "alibaba-messages",
body: {
schema: Schema.Struct({
...AnthropicMessages.AnthropicMessagesBody.fields,
thinking: Schema.optional(Schema.Struct({ type: Schema.String, budget_tokens: Schema.optional(Schema.Int) })),
}),
from: Effect.fn("AlibabaMessages.fromRequest")(function* (req) {
const opts = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Options))(req.providerOptions ?? {})
// Model Studio accepts enabled thinking without Anthropic's mandatory token budget.
return {
...(yield* AnthropicMessages.protocol.body.from(
LLMRequest.update(req, {
providerOptions: { ...req.providerOptions, thinking: undefined },
}),
)),
thinking:
opts.thinking === undefined
? undefined
: {
type: opts.thinking.type,
budget_tokens: opts.thinking.budgetTokens ?? opts.thinking.budget_tokens,
},
}
}),
},
stream: AnthropicMessages.protocol.stream,
})
export * as AlibabaMessages from "./alibaba-messages.js"
@@ -1,98 +0,0 @@
import { Effect, Schema } from "effect"
import { Protocol } from "../route/protocol.js"
import { OpenResponses } from "./open-responses.js"
import { JsonObject, optionalArray, ProviderShared } from "./shared.js"
import { OpenResponsesOptions } from "./utils/open-responses-options.js"
import { ResponsesHostedTools } from "./utils/responses-hosted-tools.js"
const Options = Schema.Struct({
reasoningEffort: OpenResponsesOptions.Options.fields.reasoningEffort,
enableThinking: Schema.optional(Schema.Boolean),
store: OpenResponsesOptions.Options.fields.store,
previousResponseId: Schema.optional(Schema.String),
conversation: Schema.optional(Schema.String),
})
export type OptionsInput = typeof Options.Type
const NativeTool = Schema.Struct({ type: Schema.Literals(["web_search", "web_extractor", "code_interpreter"]) })
const WebExtractorItem = Schema.StructWithRest(
Schema.Struct({
type: Schema.Literal("web_extractor_call"),
id: Schema.String,
urls: Schema.optional(Schema.Array(Schema.String)),
goal: Schema.optional(Schema.String),
}),
[JsonObject],
)
const Body = Schema.Struct({
...OpenResponses.coreFields,
input: Schema.Array(Schema.Union([OpenResponses.InputItem, WebExtractorItem])),
tools: optionalArray(Schema.Union([OpenResponses.Tool, NativeTool])),
enable_thinking: Options.fields.enableThinking,
previous_response_id: Options.fields.previousResponseId,
conversation: Options.fields.conversation,
stream: Schema.Literal(true),
})
const adapter = {
id: "alibaba-responses",
name: "Alibaba Responses",
nativeTool: (native) => ProviderShared.validateWith(Schema.decodeUnknownEffect(NativeTool))(native.alibaba),
restoreHostedToolItem: (item: unknown) => (Schema.is(WebExtractorItem)(item) ? item : undefined),
} satisfies OpenResponses.ProviderAdapter
const tools = {
web_search_call: { name: "web_search", input: (item) => item.action ?? {} },
code_interpreter_call: { name: "code_interpreter", input: (item) => ({ code: item.code }) },
} satisfies ResponsesHostedTools.Definitions
export const protocol = Protocol.make({
id: adapter.id,
body: {
schema: Body,
from: Effect.fn("AlibabaResponses.fromRequest")(function* (req) {
const opts = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Options))(req.providerOptions ?? {})
const body = yield* OpenResponses.fromRequestWithAdapter(req, adapter)
const choice = body.tool_choice
return yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Body))({
...body,
enable_thinking: opts.enableThinking,
previous_response_id: opts.previousResponseId,
conversation: opts.conversation,
// Model Studio expresses named selection through allowed_tools.
tool_choice:
typeof choice === "object" && choice.type === "function"
? { type: "allowed_tools" as const, mode: "required" as const, tools: [choice] }
: choice,
})
}),
},
stream: {
event: OpenResponses.protocol.stream.event,
initial: (req) => OpenResponses.initial(req, adapter),
step: (state, input) =>
Effect.gen(function* () {
const event = OpenResponses.normalize(state, input)
if (event.type !== "response.output_item.done" || !event.item) return yield* OpenResponses.step(state, event)
if (event.item.type === "web_extractor_call") {
const item = yield* Schema.decodeUnknownEffect(WebExtractorItem)(event.item).pipe(
Effect.mapError((cause) =>
ProviderShared.eventError(
adapter.id,
"Alibaba returned an invalid web extraction item",
ProviderShared.encodeJson(event),
cause,
),
),
)
return yield* ResponsesHostedTools.onDone(state, item, {
web_extractor_call: { name: "web_extractor", input: () => ({ urls: item.urls, goal: item.goal }) },
})
}
if (ResponsesHostedTools.isItem(event.item, tools))
return yield* ResponsesHostedTools.onDone(state, event.item, tools)
return yield* OpenResponses.step(state, event)
}),
terminal: OpenResponses.terminal,
},
})
export * as AlibabaResponses from "./alibaba-responses.js"
@@ -18,6 +18,7 @@ const WebSocketResponseCreate = Schema.StructWithRest(Schema.Struct({ type: Sche
])
const decodeMessage = ProviderShared.validateWith(Schema.decodeUnknownEffect(WebSocketResponseCreate))
const encodeMessage = Schema.encodeSync(Schema.fromJsonString(WebSocketResponseCreate))
const decodeEvent = Schema.decodeUnknownEffect(OpenResponses.protocol.stream.event)
export interface Options {
readonly id: string
@@ -26,7 +27,6 @@ export interface Options {
readonly enabled?: (url: string) => boolean
readonly url?: (url: string) => string
readonly headers?: (headers: Headers.Headers) => Headers.Headers
readonly continuation?: OpenResponsesContinuation.Shape
}
export interface Prepared {
@@ -60,7 +60,7 @@ const driver = (options: Options, body: string): WebSocketChannelDriver => {
}),
observe: (_create, frame) =>
Effect.gen(function* () {
const event = yield* OpenResponses.decodeChannelEvent(frame).pipe(
const event = yield* decodeEvent(frame).pipe(
Effect.mapError((cause) =>
ProviderShared.eventError(options.id, `Invalid ${options.name} WebSocket event`, frame, cause),
),
@@ -163,7 +163,6 @@ export const transport = <Body>(options: Options): Transport<Body, Prepared, str
request: create.request,
message: create.message,
base,
continuation: options.continuation,
}),
}
})
@@ -6,6 +6,7 @@ import { OpenResponses } from "./open-responses.js"
const PROTOCOL = "open-responses.websocket.v1"
const VERSION = 1
const decodeEvent = Schema.decodeUnknownEffect(OpenResponses.protocol.stream.event)
interface CheckpointValue {
readonly version: typeof VERSION
@@ -14,19 +15,12 @@ interface CheckpointValue {
readonly output: ReadonlyArray<unknown>
}
/**
* Fields to send next to `previous_response_id` on an incremental step, or undefined to send the step in full.
* Whether omitted fields carry over from the continued response is provider behavior the route must know.
*/
export type Shape = (request: Readonly<Record<string, unknown>>) => Readonly<Record<string, unknown>> | undefined
export interface DriverInput {
readonly id: string
readonly name: string
readonly request: Readonly<Record<string, unknown>>
readonly message: string
readonly base: WebSocketChannelDriver
readonly continuation?: Shape
}
const checkpointValue = (checkpoint: ChannelCheckpoint | undefined): CheckpointValue | undefined => {
@@ -133,26 +127,22 @@ const rejected = (
export const driver = (input: DriverInput): WebSocketChannelDriver => {
const { previous_response_id: _previousResponseID, ...request } = input.request
const shape = input.continuation ?? ((fields: Readonly<Record<string, unknown>>) => fields)
let output: OpenResponses.StreamItem[] = []
return {
create: (checkpoint) =>
Effect.sync(() => {
output = []
const previous = checkpointValue(checkpoint)
// Ask the route first: diffing the whole history is wasted when it declines the continuation.
const fields = previous ? shape(request) : undefined
const delta = previous && fields ? incremental(request, previous) : undefined
if (!previous || !fields || !delta)
return { message: ProviderShared.encodeJson(request), mode: "full" as const }
const delta = previous ? incremental(request, previous) : undefined
if (!previous || !delta) return { message: ProviderShared.encodeJson(request), mode: "full" as const }
return {
message: ProviderShared.encodeJson({ ...fields, input: delta, previous_response_id: previous.responseID }),
message: ProviderShared.encodeJson({ ...request, input: delta, previous_response_id: previous.responseID }),
mode: "incremental" as const,
}
}),
observe: (create, frame) =>
Effect.gen(function* () {
const event = yield* OpenResponses.decodeChannelEvent(frame).pipe(
const event = yield* decodeEvent(frame).pipe(
Effect.mapError((cause) =>
ProviderShared.eventError(input.id, `Invalid ${input.name} WebSocket event`, frame, cause),
),
@@ -163,15 +153,6 @@ export const driver = (input: DriverInput): WebSocketChannelDriver => {
const rejection = code(event)
if (rejection === "previous_response_not_found") return rejected(observation, "retry-full")
if (rejection === "websocket_connection_limit_reached") return rejected(observation, "rotate-and-retry-full")
// Only the continuation distinguishes an incremental send from a full one, so an unclassified
// invalid request there is retried full; Codex reports a stale previous_response_id that way, with
// no code. Classified failures such as context overflow keep their runner-owned recovery.
if (
create.mode === "incremental" &&
observation.error.reason._tag === "InvalidRequest" &&
observation.error.reason.classification === undefined
)
return rejected(observation, "retry-full")
}
if (observation.type !== "completed") return observation
// A trigger installs a different context window. Clear the append baseline, retaining the socket.
@@ -191,7 +172,7 @@ export const driver = (input: DriverInput): WebSocketChannelDriver => {
responseID,
request,
// Completion can re-encrypt reasoning. Callers replay the item already emitted by output_item.done.
output: event.response?.output?.length
output: event.response?.output
? event.response.output.map((item) =>
item.type === "reasoning" && item.id !== undefined
? (output.find((done) => done.type === item.type && done.id === item.id) ?? item)
@@ -205,4 +186,4 @@ export const driver = (input: DriverInput): WebSocketChannelDriver => {
}
}
export * as OpenResponsesContinuation from "./open-responses-continuation.js"
export const OpenResponsesContinuation = { driver } as const
+21 -90
View File
@@ -1,4 +1,4 @@
import { Effect, Option, Schema, SchemaGetter } from "effect"
import { Effect, Option, Schema } from "effect"
import type { Content } from "@opencode/schema/tool"
import { HttpTransport } from "../route/transport/index.js"
import { Protocol } from "../route/protocol.js"
@@ -86,7 +86,6 @@ export const OpenResponsesReasoningItem = Schema.Struct({
id: Schema.optionalKey(Schema.String),
summary: Schema.Array(OpenResponsesReasoningSummaryText),
encrypted_content: optionalNull(Schema.String),
provider_metadata: Schema.optional(JsonObject),
})
const OpenResponsesWebSearchCall = Schema.StructWithRest(
@@ -183,7 +182,6 @@ export const InputItem = Schema.Union([
content: Schema.Array(OpenResponsesOutputText),
phase: Schema.optionalKey(MessagePhase),
status: Schema.optional(Schema.String),
provider_metadata: Schema.optional(JsonObject),
}),
OpenResponsesReasoningItem,
Schema.Struct({
@@ -193,7 +191,6 @@ export const InputItem = Schema.Union([
name: Schema.String,
namespace: Schema.optional(Schema.String),
arguments: Schema.String,
provider_metadata: Schema.optional(JsonObject),
}),
Schema.Struct({
type: Schema.tag("function_call_output"),
@@ -226,7 +223,6 @@ type OpenResponsesReasoningInput = {
id?: string
summary: Array<{ type: "summary_text"; text: string }>
encrypted_content?: string | null
provider_metadata?: Record<string, unknown>
}
export const Tool = Schema.Struct({
type: Schema.tag("function"),
@@ -329,8 +325,9 @@ export const StreamItem = Schema.StructWithRest(
export type StreamItem = Schema.Schema.Type<typeof StreamItem>
export type OutputItem = StreamItem & { readonly id: string }
// Responses-compatible providers put streaming error details at the top level or
// under `error`, and response failures under `response.error`. Accept all three shapes.
// The Responses schema puts streaming error details at the top level and
// response failures under `response.error`. WebSocket failures use an
// event-level `error` envelope, so accept all three shapes here.
// https://www.openresponses.org/specification
const OpenResponsesErrorPayload = Schema.Struct({
type: optionalNull(Schema.String),
@@ -404,47 +401,13 @@ export const Event = Schema.StructWithRest(
headers: Schema.optional(Schema.Unknown),
}),
[Schema.Record(Schema.String, Schema.Unknown)],
).pipe(
Schema.decode({
decode: SchemaGetter.transform((event) => {
if (event.type !== "error" || event.error != null) return event
const { code, message, param, ...rest } = event
if (code === undefined && message === undefined && param === undefined) return event
// Flat errors (for example, Meta's) can also arrive through generic Responses endpoints.
return { ...rest, error: { code, message, param } }
}),
encode: SchemaGetter.passthrough(),
}),
)
export type Event = Schema.Schema.Type<typeof Event>
export type NormalizedEvent = Event & { readonly item?: OutputItem | null }
const decodeEventValue = Schema.decodeUnknownEffect(Event)
const decodeFrame = Schema.decodeUnknownEffect(ProviderShared.Json)
/**
* Decodes one WebSocket frame. xAI answers a rejected `response.create` with `{ "error": { "message", "type" } }` and no
* event type; that envelope reads as an error event so the failure classifies instead of failing decoding.
*/
export const decodeChannelEvent = (frame: string) =>
decodeFrame(frame).pipe(
Effect.flatMap((value) =>
decodeEventValue(
ProviderShared.isRecord(value) && value.type === undefined && ProviderShared.isRecord(value.error)
? { ...value, type: "error" }
: value,
),
),
)
export interface ProviderAdapter {
readonly id: string
readonly name: string
/** Replay opaque gateway continuation state only for adapters that own this extension. */
readonly preserveProviderMetadata?: boolean
readonly nativeTool?: (
native: NonNullable<ToolDefinition["native"]>,
) => Effect.Effect<{ readonly type: string }, AIError>
readonly lowerMedia?: (input: {
readonly part: MediaPart
readonly media: ProviderShared.NormalizedMedia
@@ -461,7 +424,6 @@ export interface ParserState {
readonly id: string
readonly name: string
readonly providerMetadataKey: string
readonly preserveProviderMetadata: boolean
readonly tools: ToolStream.State<string>
readonly hasFunctionCall: boolean
readonly lifecycle: Lifecycle.State
@@ -521,16 +483,7 @@ const itemID = (providerMetadata: ProviderMetadata | undefined, providerMetadata
return separator > 0 && separator < metadata.itemId.length - 1 ? metadata.itemId : undefined
}
const replayProviderMetadata = (metadata: ProviderMetadata | undefined, key: string, adapter: ProviderAdapter) => {
const value = metadata?.[key]?.providerMetadata
return adapter.preserveProviderMetadata && ProviderShared.isRecord(value) ? { provider_metadata: value } : {}
}
const lowerToolCall = (
part: ToolCallPart,
providerMetadataKey: string,
adapter: ProviderAdapter,
): OpenResponsesInputItem => {
const lowerToolCall = (part: ToolCallPart, providerMetadataKey: string): OpenResponsesInputItem => {
const id = itemID(part.providerMetadata, providerMetadataKey)
return {
type: "function_call",
@@ -539,15 +492,10 @@ const lowerToolCall = (
name: part.name,
namespace: part.namespace,
arguments: ProviderShared.encodeJson(part.input),
...replayProviderMetadata(part.providerMetadata, providerMetadataKey, adapter),
}
}
const lowerReasoning = (
part: ReasoningPart,
providerMetadataKey: string,
adapter: ProviderAdapter,
): OpenResponsesReasoningInput | undefined => {
const lowerReasoning = (part: ReasoningPart, providerMetadataKey: string): OpenResponsesReasoningInput | undefined => {
const metadata = part.providerMetadata?.[providerMetadataKey]
if (!ProviderShared.isRecord(metadata)) return undefined
const id = itemID(part.providerMetadata, providerMetadataKey)
@@ -560,7 +508,6 @@ const lowerReasoning = (
...(id === undefined ? {} : { id }),
summary: part.text.length > 0 ? [{ type: "summary_text", text: part.text }] : [],
encrypted_content: encryptedContent,
...replayProviderMetadata(part.providerMetadata, providerMetadataKey, adapter),
}
}
@@ -705,11 +652,9 @@ const lowerMessages = Effect.fn("OpenResponses.lowerMessages")(function* (
type: "message" as const,
...(group.id === undefined ? {} : { id: group.id }),
role: "assistant" as const,
// Replayed text is a finished input item, even if generation was cut short.
status: "completed",
status: metadata?.status,
content: group.parts.map((part) => ({ type: "output_text" as const, text: part.text })),
...(group.phase === undefined ? {} : { phase: group.phase }),
...replayProviderMetadata(group.parts.at(-1)?.providerMetadata, providerMetadataKey, adapter),
})),
)
content.splice(0, content.length)
@@ -730,14 +675,13 @@ const lowerMessages = Effect.fn("OpenResponses.lowerMessages")(function* (
}
if (part.type === "reasoning") {
flushText()
const reasoning = lowerReasoning(part, providerMetadataKey, adapter)
const reasoning = lowerReasoning(part, providerMetadataKey)
if (!reasoning) continue
const existing = reasoning.id === undefined ? undefined : reasoningItems[reasoning.id]
if (existing) {
existing.summary.push(...reasoning.summary)
if (typeof reasoning.encrypted_content === "string")
existing.encrypted_content = reasoning.encrypted_content
if (reasoning.provider_metadata !== undefined) existing.provider_metadata = reasoning.provider_metadata
continue
}
if (reasoning.id !== undefined) reasoningItems[reasoning.id] = reasoning
@@ -747,7 +691,7 @@ const lowerMessages = Effect.fn("OpenResponses.lowerMessages")(function* (
if (part.type === "tool-call") {
flushText()
if (part.providerExecuted === true) continue
input.push(lowerToolCall(part, providerMetadataKey, adapter))
input.push(lowerToolCall(part, providerMetadataKey))
continue
}
if (part.type === "tool-result" && part.providerExecuted === true) {
@@ -875,13 +819,11 @@ export const fromRequestWithAdapter = Effect.fn("OpenResponses.fromRequestWithAd
projected.tools.length === 0
? undefined
: yield* Effect.forEach(projected.tools, (tool) =>
tool.native !== undefined && adapter.nativeTool
? adapter.nativeTool(tool.native)
: lowerTool(
adapter.name,
tool,
ToolSchemaProjection.modelCompatibility(tool.inputSchema, toolSchemaCompatibility),
),
lowerTool(
adapter.name,
tool,
ToolSchemaProjection.modelCompatibility(tool.inputSchema, toolSchemaCompatibility),
),
),
tool_choice:
allowedToolChoice(request) ??
@@ -1086,17 +1028,8 @@ export const onReasoningDone = (state: ParserState, event: Event, itemID: string
return onReasoningDelta(state, { ...event, delta: event.text }, itemID)
}
const outputMetadata = (state: ParserState, item: OutputItem, extra?: Record<string, unknown>) =>
providerMetadata(state, {
itemId: item.id,
...extra,
...(state.preserveProviderMetadata && ProviderShared.isRecord(item.provider_metadata)
? { providerMetadata: item.provider_metadata }
: {}),
})
const reasoningMetadata = (state: ParserState, item: OutputItem) =>
outputMetadata(state, item, { reasoningEncryptedContent: item.encrypted_content ?? null })
providerMetadata(state, { itemId: item.id, reasoningEncryptedContent: item.encrypted_content ?? null })
// Responses APIs normally stream reasoning items in this order:
// `output_item.added` (reasoning) →
@@ -1159,7 +1092,7 @@ const onOutputItemAdded = (state: ParserState, event: NormalizedEvent): StepResu
}
if (item.type !== "function_call" || !item.call_id) return [state, NO_EVENTS]
if (state.tools[item.id] !== undefined) return [state, NO_EVENTS]
const metadata = outputMetadata(state, item)
const metadata = providerMetadata(state, { itemId: item.id })
const events: LLMEvent[] = []
const lifecycle = Lifecycle.stepStart(state.lifecycle, events)
return [
@@ -1282,7 +1215,7 @@ const onOutputItemDone = Effect.fn("OpenResponses.onOutputItemDone")(function* (
content.push(decoded.type === "output_text" ? decoded.text : decoded.refusal)
}
const text = content.length > 0 ? content.join("") : undefined
const metadata = outputMetadata(state, item, phase === undefined ? undefined : { phase })
const metadata = providerMetadata(state, { itemId: item.id, ...(phase === undefined ? {} : { phase }) })
const events: LLMEvent[] = []
const lifecycle = text ? Lifecycle.textStart(state.lifecycle, events, item.id, metadata) : state.lifecycle
return [
@@ -1297,11 +1230,10 @@ const onOutputItemDone = Effect.fn("OpenResponses.onOutputItemDone")(function* (
if (item.type === "function_call") {
if (!item.call_id || !item.name) return [state, NO_EVENTS] satisfies StepResult
const metadata = outputMetadata(state, item)
const pending = state.tools[item.id]
const registered = pending !== undefined
const tools = pending
? ToolStream.start(state.tools, item.id, { ...pending, providerMetadata: metadata })
const metadata = providerMetadata(state, { itemId: item.id })
const registered = state.tools[item.id] !== undefined
const tools = registered
? state.tools
: ToolStream.start(state.tools, item.id, {
id: item.call_id,
name: item.name,
@@ -1555,7 +1487,6 @@ export const initial = (request: LLMRequest, adapter: ProviderAdapter = BASE_ADA
id: adapter.id,
name: adapter.name,
providerMetadataKey: metadataKey(request.model),
preserveProviderMetadata: adapter.preserveProviderMetadata ?? false,
hasFunctionCall: false,
tools: ToolStream.empty<string>(),
lifecycle: Lifecycle.initial(),
@@ -1,72 +0,0 @@
import { Effect, Schema } from "effect"
import { Route } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { Protocol } from "../route/protocol.js"
import { HttpTransport } from "../route/transport/index.js"
import type { LLMRequest } from "../schema/index.js"
import { OpenResponses } from "./open-responses.js"
import { ProviderShared } from "./shared.js"
const ADAPTER = "organization-routes"
const adapter = {
id: ADAPTER,
name: "Organization routes",
preserveProviderMetadata: true,
} satisfies OpenResponses.ProviderAdapter
const Body = Schema.Struct({
...OpenResponses.coreFields,
store: Schema.Literal(false),
provider_options: Schema.Struct({
"openai-responses": Schema.Struct({ include: Schema.Array(Schema.String) }),
}),
stream: Schema.Literal(true),
})
const fromRequest = Effect.fn("OrganizationRoutes.fromRequest")(function* (request: LLMRequest) {
const body = yield* OpenResponses.fromRequestWithAdapter(request, adapter)
// A route can select another provider on each call. Keep full history and only
// portable options; native caching and stored IDs would exclude translated
// targets. Scope encrypted reasoning to native Responses targets so their
// stateless continuations remain replayable without blocking other protocols.
return yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Body))({
model: body.model,
input: body.input,
instructions: body.instructions,
tools: body.tools,
tool_choice: body.tool_choice,
stream: true as const,
store: false as const,
provider_options: { "openai-responses": { include: ["reasoning.encrypted_content"] } },
max_output_tokens:
body.max_output_tokens !== undefined && body.max_output_tokens > 0 ? body.max_output_tokens : undefined,
temperature: body.temperature,
top_p: body.top_p,
parallel_tool_calls: body.parallel_tool_calls,
metadata: body.metadata,
reasoning: body.reasoning?.effort === undefined ? undefined : { effort: body.reasoning.effort },
text: body.text,
})
})
export const protocol = Protocol.make({
...OpenResponses.protocol,
id: ADAPTER,
body: { schema: Body, from: fromRequest },
stream: {
...OpenResponses.protocol.stream,
initial: (request: LLMRequest) => OpenResponses.initial(request, adapter),
},
})
export const route = Route.make({
id: ADAPTER,
providerMetadataKey: ADAPTER,
protocol,
endpoint: Endpoint.path(OpenResponses.PATH),
transport: HttpTransport.sseJson.with<typeof Body.Type>(),
defaults: { providerOptions: { store: false, include: [] } },
})
export * as OrganizationRoutes from "./organization-routes.js"
-139
View File
@@ -1,139 +0,0 @@
import type { ProviderPackage } from "../provider-package.js"
import { AlibabaChat } from "../protocols/alibaba-chat.js"
import { AlibabaMessages } from "../protocols/alibaba-messages.js"
import { AlibabaResponses } from "../protocols/alibaba-responses.js"
import { AuthOptions, type AtLeastOne, type ProviderAuthOption } from "../route/auth-options.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { Framing } from "../route/framing.js"
import { ProviderConfigurationError, ProviderID, ToolDefinition, type ModelID } from "../schema/index.js"
export const id = ProviderID.make("alibaba")
export type Region =
| "ap-southeast-1"
| "cn-beijing"
| "cn-hongkong"
| "us-east-1"
| "eu-central-1"
| "ap-northeast-1"
| (string & {})
export type ChatOptionsInput = AlibabaChat.OptionsInput
export type MessagesOptionsInput = AlibabaMessages.OptionsInput
export type ResponsesOptionsInput = AlibabaResponses.OptionsInput
type Location = AtLeastOne<{
readonly region: Region
/** Overrides the selected API's complete base URL, including its version prefix. */
readonly baseURL: string
}> & { readonly workspaceID?: string }
export type Config = Location &
Omit<RouteDefaultsInput, "providerOptions"> &
ProviderAuthOption<"optional"> & {
readonly providerOptions?: ChatOptionsInput | MessagesOptionsInput | ResponsesOptionsInput
}
export type Settings<Options = ChatOptionsInput> = Location &
ProviderPackage.Settings & {
readonly apiKey?: string
readonly providerOptions?: Options
}
const hosts = new Map<string, string>([
["ap-southeast-1", "dashscope-intl.aliyuncs.com"],
["cn-beijing", "dashscope.aliyuncs.com"],
["cn-hongkong", "cn-hongkong.dashscope.aliyuncs.com"],
["us-east-1", "dashscope-us.aliyuncs.com"],
])
const chatRoute = Route.make({
id: "alibaba-chat",
provider: id,
providerMetadataKey: "alibaba",
protocol: AlibabaChat.protocol,
endpoint: Endpoint.path("/chat/completions"),
framing: Framing.sse,
})
const messagesRoute = Route.make({
id: "alibaba-messages",
provider: id,
providerMetadataKey: "alibaba",
protocol: AlibabaMessages.protocol,
endpoint: Endpoint.path("/messages"),
framing: Framing.sse,
headers: () => ({ "anthropic-version": "2023-06-01" }),
})
const responsesRoute = Route.make({
id: "alibaba-responses",
provider: id,
providerMetadataKey: "alibaba",
protocol: AlibabaResponses.protocol,
endpoint: Endpoint.path("/responses"),
framing: Framing.sse,
})
export const routes = [chatRoute, messagesRoute, responsesRoute]
export const configure = (input: Config) => {
const { apiKey: _key, auth: _auth, region, workspaceID, baseURL, ...rest } = input
const host =
region === undefined
? undefined
: workspaceID === undefined
? hosts.get(region)
: `${workspaceID}.${region}.maas.aliyuncs.com`
if (baseURL === undefined) {
if (region === undefined)
throw new ProviderConfigurationError({ provider: id, message: "Alibaba requires region or baseURL" })
if (host === undefined)
throw new ProviderConfigurationError({
provider: id,
message: `Alibaba region ${region} requires workspaceID or baseURL`,
})
}
const opts = { ...rest, auth: AuthOptions.bearer(input, ["DASHSCOPE_API_KEY", "ALIBABA_API_KEY"]) }
const common = { ...opts, endpoint: { baseURL: baseURL ?? `https://${host}/compatible-mode/v1` } }
const chat = (id: string | ModelID) =>
chatRoute.with(common).model<ChatOptionsInput>({ id, compatibility: AlibabaChat.compatibility })
const messages = (id: string | ModelID) =>
messagesRoute
.with({
...opts,
endpoint: { baseURL: baseURL ?? `https://${host}/apps/anthropic/v1` },
})
.model<MessagesOptionsInput>({ id, compatibility: { requireSignature: false } })
const responses = (id: string | ModelID) => responsesRoute.with(common).model<ResponsesOptionsInput>({ id })
return { id, model: chat, chat, messages, responses, configure }
}
export const provider = { id, configure }
export const model: ProviderPackage.Definition<Settings, ChatOptionsInput>["model"] = (id, input) =>
fromSettings(input).chat(id)
export const messagesModel: ProviderPackage.Definition<
Settings<MessagesOptionsInput>,
MessagesOptionsInput
>["model"] = (id, input) => fromSettings(input).messages(id)
export const responsesModel: ProviderPackage.Definition<
Settings<ResponsesOptionsInput>,
ResponsesOptionsInput
>["model"] = (id, input) => fromSettings(input).responses(id)
function fromSettings(input: Settings<Config["providerOptions"]>) {
const { body, ...rest } = input
return configure({ ...rest, http: body === undefined ? undefined : { body } })
}
export const webSearch = () => hostedTool("web_search", "Search the web with Alibaba's hosted search tool.")
export const webExtractor = () => hostedTool("web_extractor", "Extract web page content with Alibaba's hosted tool.")
export const codeInterpreter = () => hostedTool("code_interpreter", "Execute code with Alibaba's hosted interpreter.")
function hostedTool(type: "web_search" | "web_extractor" | "code_interpreter", description: string) {
return ToolDefinition.make({
name: type,
description,
inputSchema: { type: "object", properties: {} },
native: { alibaba: { type } },
})
}
export * as Alibaba from "./alibaba.js"
@@ -1 +0,0 @@
export { model, type Settings } from "../alibaba.js"
@@ -1,3 +0,0 @@
import type { Alibaba } from "../alibaba.js"
export { messagesModel as model } from "../alibaba.js"
export type Settings = Alibaba.Settings<Alibaba.MessagesOptionsInput>
@@ -1,3 +0,0 @@
import type { Alibaba } from "../alibaba.js"
export { responsesModel as model } from "../alibaba.js"
export type Settings = Alibaba.Settings<Alibaba.ResponsesOptionsInput>
@@ -1,10 +1,9 @@
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import type { ProviderPackage } from "../provider-package.js"
import { OpenAIChat } from "../protocols/openai-chat.js"
import { OpenResponses } from "../protocols/open-responses.js"
import { OpenAIResponses } from "../protocols/openai-responses.js"
import { BedrockAuth, type Credentials } from "../protocols/utils/bedrock-auth.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import { withOpenAIOptions, type OpenAIProviderOptionsInput } from "./openai-options.js"
export const id = ProviderID.make("amazon-bedrock")
@@ -38,10 +37,11 @@ const responsesRoute = Route.make({
id: "bedrock-mantle-responses",
provider: id,
providerMetadataKey: "mantle",
protocol: OpenResponses.protocol,
endpoint: Endpoint.path(OpenResponses.PATH),
transport: OpenResponses.httpTransport,
defaults: { providerOptions: { store: false, include: ["reasoning.encrypted_content"] } },
protocol: OpenAIResponses.protocol,
endpoint: OpenAIResponses.route.endpoint,
auth: OpenAIResponses.route.auth,
transport: OpenAIResponses.httpTransport,
defaults: OpenAIResponses.route.defaults,
})
const chatRoute = OpenAIChat.route.with({
@@ -79,12 +79,9 @@ const defaults = (input: Config) => {
export const configure = (input: Config = {}) => {
if (input.auth === "bearer" && input.apiKey === undefined && process.env.AWS_BEARER_TOKEN_BEDROCK === undefined)
throw new ProviderConfigurationError({ provider: id, message: "Amazon Bedrock Mantle bearer auth requires apiKey" })
throw new Error("Amazon Bedrock Mantle bearer auth requires apiKey")
if (input.auth === "sigv4" && input.apiKey !== undefined)
throw new ProviderConfigurationError({
provider: id,
message: "Amazon Bedrock Mantle SigV4 auth does not accept apiKey",
})
throw new Error("Amazon Bedrock Mantle SigV4 auth does not accept apiKey")
const configuredResponsesRoute = configuredRoute(responsesRoute, input)
const configuredChatRoute = configuredRoute(chatRoute, input)
const modelDefaults = defaults(input)
+3 -4
View File
@@ -1,6 +1,6 @@
import type { RouteDefaultsInput } from "../route/client.js"
import type { ProviderPackage } from "../provider-package.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import * as BedrockConverse from "../protocols/bedrock-converse.js"
import type { BedrockCredentials } from "../protocols/bedrock-converse.js"
import { BedrockAuth } from "../protocols/utils/bedrock-auth.js"
@@ -39,9 +39,8 @@ const bedrockBaseURL = (region: string) => `https://bedrock-runtime.${region}.am
const configuredRoute = (input: Config) => {
const { apiKey, auth, credentials, profile, region, baseURL, ...rest } = input
if (auth === "bearer" && apiKey === undefined && process.env.AWS_BEARER_TOKEN_BEDROCK === undefined)
throw new ProviderConfigurationError({ provider: id, message: "Amazon Bedrock bearer auth requires apiKey" })
if (auth === "sigv4" && apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Amazon Bedrock SigV4 auth does not accept apiKey" })
throw new Error("Amazon Bedrock bearer auth requires apiKey")
if (auth === "sigv4" && apiKey !== undefined) throw new Error("Amazon Bedrock SigV4 auth does not accept apiKey")
const resolvedRegion = BedrockAuth.resolveRegion(input)
return BedrockConverse.route.with({
...rest,
@@ -3,7 +3,7 @@ import { AnthropicMessages } from "../protocols/anthropic-messages.js"
import { Auth } from "../route/auth.js"
import type { ProviderAuthOption } from "../route/auth-options.js"
import type { RouteDefaultsInput } from "../route/client.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
export type AnthropicOptionsInput = AnthropicMessages.OptionsInput
export type AnthropicProviderOptionsInput = AnthropicMessages.ProviderOptionsInput
@@ -36,12 +36,8 @@ const auth = (input: ProviderAuthOption<"optional">) => {
}
export const configure = (input: Config) => {
if (!input.baseURL) throw new Error("Anthropic-compatible providers require a baseURL")
const provider = input.provider ?? "anthropic-compatible"
if (!input.baseURL)
throw new ProviderConfigurationError({
provider: ProviderID.make(provider),
message: "Anthropic-compatible providers require a baseURL",
})
const { provider: _, baseURL, apiKey: _apiKey, auth: _auth, ...rest } = input
const route = AnthropicMessages.route.with({
...rest,
@@ -65,13 +61,8 @@ export const model: ProviderPackage.Definition<Settings, AnthropicMessages.Provi
modelID,
settings,
) => {
// Read before the exclusivity check narrows a conflicting settings object to `never`.
const provider = ProviderID.make(settings.provider ?? id)
if (settings.apiKey !== undefined && settings.authToken !== undefined)
throw new ProviderConfigurationError({
provider,
message: "Anthropic-compatible apiKey cannot be combined with authToken",
})
throw new Error("Anthropic-compatible apiKey cannot be combined with authToken")
return configure({
...(settings.authToken === undefined ? { apiKey: settings.apiKey } : { auth: Auth.bearer(settings.authToken) }),
baseURL: settings.baseURL,
+2 -5
View File
@@ -2,7 +2,7 @@ import type { RouteDefaultsInput } from "../route/client.js"
import { Auth } from "../route/auth.js"
import type { ProviderAuthOption } from "../route/auth-options.js"
import type { ProviderPackage } from "../provider-package.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import { AnthropicMessages } from "../protocols/anthropic-messages.js"
import { AnthropicCompatible } from "./anthropic-compatible.js"
@@ -57,10 +57,7 @@ export const model: ProviderPackage.Definition<Settings, AnthropicMessages.Provi
settings,
) => {
if (settings.apiKey !== undefined && settings.authToken !== undefined)
throw new ProviderConfigurationError({
provider: id,
message: "Anthropic apiKey cannot be combined with authToken",
})
throw new Error("Anthropic apiKey cannot be combined with authToken")
return configure({
...(settings.authToken === undefined ? { apiKey: settings.apiKey } : { auth: Auth.bearer(settings.authToken) }),
baseURL: settings.baseURL,
+2 -2
View File
@@ -3,7 +3,7 @@ import { Auth } from "../route/auth.js"
import { type AtLeastOne, type ProviderAuthOption } from "../route/auth-options.js"
import type { Route, RouteDefaultsInput, CompactionOperations } from "../route/client.js"
import type { ProviderPackage } from "../provider-package.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import * as OpenAIChat from "../protocols/openai-chat.js"
import * as OpenAIResponses from "../protocols/openai-responses.js"
import { ProviderShared } from "../protocols/shared.js"
@@ -163,7 +163,7 @@ const config = (settings: Settings): Config => {
}
if (settings.baseURL !== undefined) return { ...common, baseURL: settings.baseURL }
if (settings.resourceName !== undefined) return { ...common, resourceName: settings.resourceName }
throw new ProviderConfigurationError({ provider: id, message: "Azure requires resourceName or baseURL" })
throw new Error("Azure requires resourceName or baseURL")
}
export const responsesModel: ProviderPackage.Definition<
@@ -5,7 +5,7 @@ import { Auth } from "../route/auth.js"
import type { AtLeastOne, ProviderAuthOption } from "../route/auth-options.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import type { OpenAIProviderOptionsInput } from "./openai-options.js"
export const id = ProviderID.make("cloudflare-ai-gateway")
@@ -35,11 +35,7 @@ export type Settings = ProviderPackage.Settings &
export const baseURL = (input: GatewayURL) => {
if (input.baseURL) return input.baseURL
if (!input.accountId)
throw new ProviderConfigurationError({
provider: id,
message: "CloudflareAIGateway.configure requires accountId unless baseURL is supplied",
})
if (!input.accountId) throw new Error("CloudflareAIGateway.configure requires accountId unless baseURL is supplied")
return `https://gateway.ai.cloudflare.com/v1/${encodeURIComponent(input.accountId)}/${encodeURIComponent(input.gatewayId?.trim() || "default")}/compat`
}
@@ -3,7 +3,7 @@ import { OpenAIChat } from "../protocols/openai-chat.js"
import { AuthOptions, type AtLeastOne, type ProviderAuthOption } from "../route/auth-options.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import type { OpenAIProviderOptionsInput } from "./openai-options.js"
export const id = ProviderID.make("cloudflare-workers-ai")
@@ -28,11 +28,7 @@ export type Settings = ProviderPackage.Settings &
export const baseURL = (input: WorkersAIURL) => {
if (input.baseURL) return input.baseURL
if (!input.accountId)
throw new ProviderConfigurationError({
provider: id,
message: "CloudflareWorkersAI.configure requires accountId unless baseURL is supplied",
})
if (!input.accountId) throw new Error("CloudflareWorkersAI.configure requires accountId unless baseURL is supplied")
return `https://api.cloudflare.com/client/v4/accounts/${encodeURIComponent(input.accountId)}/ai/v1`
}
@@ -2,7 +2,7 @@ import type { ProviderPackage } from "../provider-package.js"
import { OpenAIChat } from "../protocols/openai-chat.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import { GoogleVertexShared } from "./google-vertex-shared.js"
import type { OpenAIProviderOptionsInput } from "./openai-options.js"
@@ -37,8 +37,7 @@ const route = Route.make({
export const routes = [route]
const configuredRoute = (input: Config) => {
if ("apiKey" in input && input.apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Google Vertex Chat does not support API keys" })
if ("apiKey" in input && input.apiKey !== undefined) throw new Error("Google Vertex Chat does not support API keys")
const {
accessToken: _accessToken,
auth: _auth,
@@ -75,8 +74,7 @@ export const provider = {
}
export const model: ProviderPackage.Definition<Settings, OpenAIProviderOptionsInput>["model"] = (modelID, settings) => {
if (settings.apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Google Vertex Chat does not support API keys" })
if (settings.apiKey !== undefined) throw new Error("Google Vertex Chat does not support API keys")
return configure({
accessToken: settings.accessToken,
baseURL: settings.baseURL,
@@ -5,7 +5,7 @@ import { Auth } from "../route/auth.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { Protocol } from "../route/protocol.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import { GoogleVertexShared } from "./google-vertex-shared.js"
export type AnthropicOptionsInput = AnthropicMessages.OptionsInput
@@ -67,7 +67,7 @@ export const routes = [route]
const configuredRoute = (input: Config) => {
if ("apiKey" in input && input.apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Google Vertex Messages does not support API keys" })
throw new Error("Google Vertex Messages does not support API keys")
const {
accessToken: _accessToken,
auth: _auth,
@@ -107,8 +107,7 @@ export const model: ProviderPackage.Definition<Settings, AnthropicMessages.Provi
modelID,
settings,
) => {
if (settings.apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Google Vertex Messages does not support API keys" })
if (settings.apiKey !== undefined) throw new Error("Google Vertex Messages does not support API keys")
return configure({
accessToken: settings.accessToken,
baseURL: settings.baseURL,
@@ -2,7 +2,7 @@ import type { ProviderPackage } from "../provider-package.js"
import { OpenResponses } from "../protocols/open-responses.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { ProviderConfigurationError, ProviderID, type ModelID } from "../schema/index.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import { GoogleVertexShared } from "./google-vertex-shared.js"
import type { OpenResponsesProviderOptionsInput } from "./open-responses-options.js"
@@ -39,7 +39,7 @@ export const routes = [route]
const configuredRoute = (input: Config) => {
if ("apiKey" in input && input.apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Google Vertex Responses does not support API keys" })
throw new Error("Google Vertex Responses does not support API keys")
const {
accessToken: _accessToken,
auth: _auth,
@@ -79,8 +79,7 @@ export const model: ProviderPackage.Definition<Settings, OpenResponsesProviderOp
modelID,
settings,
) => {
if (settings.apiKey !== undefined)
throw new ProviderConfigurationError({ provider: id, message: "Google Vertex Responses does not support API keys" })
if (settings.apiKey !== undefined) throw new Error("Google Vertex Responses does not support API keys")
return configure({
accessToken: settings.accessToken,
baseURL: settings.baseURL,
@@ -1,10 +1,8 @@
import type { AnyAuthClient } from "google-auth-library"
import { Effect, Redacted } from "effect"
import { Auth, MissingCredentialError } from "../route/auth.js"
import { ProviderConfigurationError, ProviderID } from "../schema/index.js"
const SCOPE = "https://www.googleapis.com/auth/cloud-platform"
const id = ProviderID.make("google-vertex")
export type OAuthOptions =
| { readonly accessToken?: string; readonly auth?: never }
@@ -37,18 +35,12 @@ export const host = (location: string) => {
export const requireProject = (value: string | undefined) => {
if (value) return value
throw new ProviderConfigurationError({
provider: id,
message: "Google Vertex requires a project when baseURL is not configured",
})
throw new Error("Google Vertex requires a project when baseURL is not configured")
}
export const apiKey = (input: ApiKeyOptions) => {
if (input.apiKey !== undefined && (input.accessToken !== undefined || input.auth !== undefined))
throw new ProviderConfigurationError({
provider: id,
message: "Google Vertex apiKey cannot be combined with accessToken or auth",
})
throw new Error("Google Vertex apiKey cannot be combined with accessToken or auth")
if (input.accessToken !== undefined || input.auth !== undefined) return undefined
return input.apiKey ?? process.env.GOOGLE_VERTEX_API_KEY
}
@@ -76,10 +68,7 @@ const adc = (project?: string) => {
export const oauth = (input: OAuthOptions, project?: string) => {
if (input.accessToken !== undefined && input.auth !== undefined)
throw new ProviderConfigurationError({
provider: id,
message: "Google Vertex accessToken cannot be combined with auth",
})
throw new Error("Google Vertex accessToken cannot be combined with auth")
if (input.auth) return input.auth
if (input.accessToken !== undefined) return Auth.bearer(input.accessToken)
return adc(project)
+3 -9
View File
@@ -6,7 +6,7 @@ import { Auth } from "../route/auth.js"
import { Route, type RouteDefaultsInput } from "../route/client.js"
import { Endpoint } from "../route/endpoint.js"
import { Framing } from "../route/framing.js"
import { ProviderConfigurationError, ProviderID, type LLMRequest, type ModelID } from "../schema/index.js"
import { ProviderID, type LLMRequest, type ModelID } from "../schema/index.js"
import { GoogleVertexShared } from "./google-vertex-shared.js"
export interface GeminiOptionsInput extends Gemini.OptionsInput {
@@ -93,10 +93,7 @@ const configuredRoute = (input: Config, modelID: string | ModelID) => {
const apiKey = GoogleVertexShared.apiKey(input)
const endpointModel = String(modelID).startsWith("endpoints/")
if (apiKey !== undefined && endpointModel)
throw new ProviderConfigurationError({
provider: id,
message: "Google Vertex tuned models do not support Express Mode API keys",
})
throw new Error("Google Vertex tuned models do not support Express Mode API keys")
const location = GoogleVertexShared.location(inputLocation, "us-central1")
const project = GoogleVertexShared.project(inputProject)
const endpoint =
@@ -126,10 +123,7 @@ export const provider = {
}
export const model: ProviderPackage.Definition<Settings, GeminiProviderOptionsInput>["model"] = (modelID, settings) => {
if (settings.apiKey !== undefined && settings.accessToken !== undefined)
throw new ProviderConfigurationError({
provider: id,
message: "Google Vertex apiKey cannot be combined with accessToken or auth",
})
throw new Error("Google Vertex apiKey cannot be combined with accessToken or auth")
return configure({
...(settings.apiKey === undefined ? { accessToken: settings.accessToken } : { apiKey: settings.apiKey }),
baseURL: settings.baseURL,
-1
View File
@@ -1,4 +1,3 @@
export * as Alibaba from "./alibaba.js"
export * as Anthropic from "./anthropic.js"
export * as AnthropicCompatible from "./anthropic-compatible.js"
export * as AmazonBedrock from "./amazon-bedrock.js"
@@ -1,57 +0,0 @@
import type { ProviderPackage } from "../provider-package.js"
import { OrganizationRoutes } from "../protocols/organization-routes.js"
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
import type { RouteDefaultsInput } from "../route/client.js"
import { ProviderID, type ModelID } from "../schema/index.js"
import type { OpenResponsesProviderOptionsInput } from "./open-responses-options.js"
export type Options = Pick<
OpenResponsesProviderOptionsInput,
"reasoningEffort" | "textVerbosity" | "parallelToolCalls" | "metadata" | "allowedTools"
>
export const id = ProviderID.make("organization-routes")
export type Config = RouteDefaultsInput &
ProviderAuthOption<"optional"> & {
readonly provider?: string
readonly baseURL: string
readonly providerOptions?: Options
}
export interface Settings extends ProviderPackage.Settings {
readonly apiKey?: string
readonly baseURL: string
readonly provider?: string
readonly providerOptions?: Options
}
export const routes = [OrganizationRoutes.route]
export const configure = (input: Config) => {
const provider = input.provider ?? "organization-routes"
const { provider: _, baseURL, apiKey: _apiKey, auth: _auth, ...rest } = input
const route = OrganizationRoutes.route.with({
...rest,
provider,
endpoint: { baseURL },
auth: AuthOptions.bearer(input, []),
})
return {
id: ProviderID.make(provider),
model: (modelID: string | ModelID) => route.model<Options>({ id: modelID }),
configure,
}
}
export const provider = { id, configure }
export const model: ProviderPackage.Definition<Settings, Options>["model"] = (modelID, settings) =>
configure({
apiKey: settings.apiKey,
baseURL: settings.baseURL,
headers: settings.headers === undefined ? undefined : { ...settings.headers },
http: settings.body === undefined ? undefined : { body: { ...settings.body } },
provider: settings.provider,
providerOptions: settings.providerOptions,
}).model(modelID)
-4
View File
@@ -41,10 +41,6 @@ const responsesRoute = Route.make({
id: "openai-responses",
name: "xAI Responses",
rotateAfterMs: RESPONSES_WEBSOCKET_ROTATE_AFTER_MS,
// xAI continues a chain only from stored responses: with `store: false` (the route default) `previous_response_id`
// fails with "Response with id=… not found", so those steps are sent in full over the reused connection. It also
// rejects `instructions` next to `previous_response_id` and keeps the instructions of the response it continues.
continuation: ({ instructions: _instructions, ...request }) => (request.store === false ? undefined : request),
}),
defaults: { providerOptions: { store: false, include: ["reasoning.encrypted_content"] } },
})
+1 -5
View File
@@ -23,7 +23,6 @@ import {
LanguageModel,
LLMEvent,
InvalidProviderOutputError,
ProviderConfigurationError,
ProviderID,
mergeGenerationOptions,
mergeHttpOptions,
@@ -129,10 +128,7 @@ const makeRouteLanguageModel = <Options extends ProviderOptions, Compact extends
const provider = route.provider ?? ("provider" in mapped ? mapped.provider : undefined)
if (!provider) throw new Error(`Route.model(${route.id}) requires a provider`)
if (!endpointBaseURL(route.endpoint))
throw new ProviderConfigurationError({
provider: ProviderID.make(provider),
message: `Route.model(${route.id}) requires an endpoint baseURL — configure it on the route first`,
})
throw new Error(`Route.model(${route.id}) requires an endpoint baseURL — configure it on the route first`)
return LanguageModel.make<Options, Compact>({
...mapped,
provider,
+2 -5
View File
@@ -115,11 +115,8 @@ const waitOpen = (ws: globalThis.WebSocket, input: WebSocketRequest) => {
}
const onAbort = () => {
cleanup()
if (ws.readyState === globalThis.WebSocket.CLOSED || ws.readyState === globalThis.WebSocket.CLOSING) return
// Node's ws reports an aborted handshake as an error event on the next tick; with no listener left
// after cleanup, EventEmitter would throw it as an uncaught exception.
ws.addEventListener("error", () => {}, { once: true })
ws.close(1000)
if (ws.readyState !== globalThis.WebSocket.CLOSED && ws.readyState !== globalThis.WebSocket.CLOSING)
ws.close(1000)
}
const onOpen = () => {
cleanup()
-13
View File
@@ -50,19 +50,6 @@ export class UnsupportedOperationError extends Schema.TaggedError<UnsupportedOpe
route: Schema.optional(RouteID),
}) {}
/**
* Provider settings that are missing, conflicting, or unsupported, such as
* Azure without `resourceName` or `baseURL`. Thrown synchronously while a
* provider facade or package entrypoint configures a model, before any
* request exists, so it is not an `AIError` reason.
*/
export class ProviderConfigurationError extends Schema.TaggedError<ProviderConfigurationError>(
"AI.Error.ProviderConfiguration",
)("ProviderConfiguration", {
provider: ProviderID,
message: Schema.String,
}) {}
export class NoRouteError extends Schema.TaggedError<NoRouteError>("AI.Error.NoRoute")("NoRoute", {
...ReasonFields,
route: RouteID,
@@ -1,35 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-chat",
"provider:alibaba",
"protocol:alibaba-chat",
"region:ap-southeast-1",
"thinking",
"usage"
],
"name": "alibaba-chat/qwen-3-7-plus-streams-thinking-disabled",
"recordedAt": "2026-09-08T03:10:42.782Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.7-plus\",\"messages\":[{\"role\":\"user\",\"content\":\"What is 173 multiplied by 219? Reply with only the final integer.\"}],\"stream\":true,\"stream_options\":{\"include_usage\":true},\"max_completion_tokens\":4096,\"enable_thinking\":false}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=utf-8"
},
"body": "data: {\"model\":\"qwen3.7-plus\",\"id\":\"chatcmpl-047bcb67-b193-9a4f-9d77-ea0325be6d7c\",\"created\":1788837041,\"object\":\"chat.completion.chunk\",\"usage\":null,\"choices\":[{\"logprobs\":null,\"index\":0,\"delta\":{\"content\":\"\",\"role\":\"assistant\"},\"finish_reason\":null}]}\n\ndata: {\"model\":\"qwen3.7-plus\",\"id\":\"chatcmpl-047bcb67-b193-9a4f-9d77-ea0325be6d7c\",\"choices\":[{\"delta\":{\"content\":\"3\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837041,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.7-plus\",\"id\":\"chatcmpl-047bcb67-b193-9a4f-9d77-ea0325be6d7c\",\"choices\":[{\"delta\":{\"content\":\"7887\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837041,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.7-plus\",\"id\":\"chatcmpl-047bcb67-b193-9a4f-9d77-ea0325be6d7c\",\"choices\":[{\"delta\":{\"content\":\"\"},\"index\":0,\"finish_reason\":\"stop\",\"logprobs\":null}],\"created\":1788837041,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"choices\":[],\"created\":1788837041,\"id\":\"chatcmpl-047bcb67-b193-9a4f-9d77-ea0325be6d7c\",\"model\":\"qwen3.7-plus\",\"object\":\"chat.completion.chunk\",\"usage\":{\"completion_tokens\":5,\"prompt_tokens\":32,\"prompt_tokens_details\":{\"cached_tokens\":0,\"text_tokens\":32},\"total_tokens\":37}}\n\ndata: [DONE]\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,35 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-chat",
"provider:alibaba",
"protocol:alibaba-chat",
"region:ap-southeast-1",
"tool",
"tool-choice"
],
"name": "alibaba-chat/qwen-3-8-max-obeys-named-tool-choice",
"recordedAt": "2026-09-08T03:10:56.576Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":\"Find the current weather in Paris.\"}],\"tools\":[{\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"description\":\"Get weather in a city\",\"parameters\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\",\"enum\":[\"Paris\"]}},\"required\":[\"city\"]}}}],\"tool_choice\":{\"type\":\"function\",\"function\":{\"name\":\"get_weather\"}},\"stream\":true,\"stream_options\":{\"include_usage\":true},\"reasoning_effort\":\"none\",\"max_completion_tokens\":4096}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=utf-8"
},
"body": "data: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null,\"choices\":[{\"logprobs\":null,\"index\":0,\"delta\":{\"content\":\"\",\"role\":\"assistant\",\"tool_calls\":[{\"index\":0,\"id\":\"call_4eb823cb28d141c8befbb331\",\"type\":\"function\",\"function\":{\"name\":\"get_weather\",\"arguments\":\"\"}}]},\"finish_reason\":null}]}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{\"content\":\"\",\"tool_calls\":[{\"index\":0,\"id\":\"\",\"type\":\"function\",\"function\":{\"arguments\":\"\"}}]},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{\"content\":\"\",\"tool_calls\":[{\"type\":\"function\",\"index\":0,\"function\":{\"arguments\":\"{\\\"city\\\": \\\"Paris\"}}]},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{\"content\":\"\",\"tool_calls\":[{\"type\":\"function\",\"index\":0,\"function\":{\"arguments\":\"\\\"\"}}]},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{\"content\":\"\",\"tool_calls\":[{\"type\":\"function\",\"index\":0,\"function\":{\"arguments\":\"}\"}}]},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{\"content\":\"\",\"tool_calls\":[{\"type\":\"function\",\"index\":0,\"function\":{\"arguments\":\"\"}}]},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{\"tool_calls\":[{\"function\":{\"arguments\":\"\"},\"index\":0,\"id\":null,\"type\":\"function\"}],\"content\":\"\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"choices\":[{\"delta\":{},\"index\":0,\"finish_reason\":\"stop\",\"logprobs\":null}],\"created\":1788837055,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"choices\":[],\"created\":1788837055,\"id\":\"chatcmpl-681dfa5a-ae08-98da-ba0d-4b1d0d364b6b\",\"model\":\"qwen3.8-max\",\"object\":\"chat.completion.chunk\",\"usage\":{\"completion_tokens\":19,\"prompt_tokens\":288,\"prompt_tokens_details\":{\"cached_tokens\":0,\"text_tokens\":288},\"total_tokens\":307}}\n\ndata: [DONE]\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
@@ -1,34 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-chat",
"provider:alibaba",
"protocol:alibaba-chat",
"region:ap-southeast-1",
"structured-output"
],
"name": "alibaba-chat/qwen-3-8-max-returns-a-json-object",
"recordedAt": "2026-09-08T03:11:29.462Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":\"Return a JSON object with one key \\\"city\\\" set to the capital city of France.\"}],\"stream\":true,\"stream_options\":{\"include_usage\":true},\"reasoning_effort\":\"none\",\"max_completion_tokens\":1024,\"response_format\":{\"type\":\"json_object\"}}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=utf-8"
},
"body": "data: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-de34af77-7b1e-9999-a48f-e564c03d4b6b\",\"created\":1788837088,\"object\":\"chat.completion.chunk\",\"usage\":null,\"choices\":[{\"logprobs\":null,\"index\":0,\"delta\":{\"content\":\"\",\"role\":\"assistant\"},\"finish_reason\":null}]}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-de34af77-7b1e-9999-a48f-e564c03d4b6b\",\"choices\":[{\"delta\":{\"content\":\"{\\\"\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837088,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-de34af77-7b1e-9999-a48f-e564c03d4b6b\",\"choices\":[{\"delta\":{\"content\":\"city\\\":\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837088,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-de34af77-7b1e-9999-a48f-e564c03d4b6b\",\"choices\":[{\"delta\":{\"content\":\" \\\"Paris\\\"}\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788837088,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-de34af77-7b1e-9999-a48f-e564c03d4b6b\",\"choices\":[{\"delta\":{\"content\":\"\"},\"index\":0,\"finish_reason\":\"stop\",\"logprobs\":null}],\"created\":1788837088,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"choices\":[],\"created\":1788837088,\"id\":\"chatcmpl-de34af77-7b1e-9999-a48f-e564c03d4b6b\",\"model\":\"qwen3.8-max\",\"object\":\"chat.completion.chunk\",\"usage\":{\"completion_tokens\":6,\"prompt_tokens\":32,\"prompt_tokens_details\":{\"cached_tokens\":0,\"text_tokens\":32},\"total_tokens\":38}}\n\ndata: [DONE]\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,36 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-chat",
"provider:alibaba",
"protocol:alibaba-chat",
"region:ap-southeast-1",
"text",
"reasoning",
"usage"
],
"name": "alibaba-chat/qwen-3-8-max-streams-none-effort",
"recordedAt": "2026-09-08T03:09:27.584Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/chat/completions",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":\"What is 173 multiplied by 219? Reply with only the final integer.\"}],\"stream\":true,\"stream_options\":{\"include_usage\":true},\"reasoning_effort\":\"none\",\"max_completion_tokens\":4096}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=utf-8"
},
"body": "data: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-d8fbc8fe-4d5a-9dc9-84c6-ed0c7b9dd3bb\",\"created\":1788836967,\"object\":\"chat.completion.chunk\",\"usage\":null,\"choices\":[{\"logprobs\":null,\"index\":0,\"delta\":{\"content\":\"\",\"role\":\"assistant\"},\"finish_reason\":null}]}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-d8fbc8fe-4d5a-9dc9-84c6-ed0c7b9dd3bb\",\"choices\":[{\"delta\":{\"content\":\"3\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788836967,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-d8fbc8fe-4d5a-9dc9-84c6-ed0c7b9dd3bb\",\"choices\":[{\"delta\":{\"content\":\"78\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788836967,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-d8fbc8fe-4d5a-9dc9-84c6-ed0c7b9dd3bb\",\"choices\":[{\"delta\":{\"content\":\"87\"},\"index\":0,\"finish_reason\":null,\"logprobs\":null}],\"created\":1788836967,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"model\":\"qwen3.8-max\",\"id\":\"chatcmpl-d8fbc8fe-4d5a-9dc9-84c6-ed0c7b9dd3bb\",\"choices\":[{\"delta\":{\"content\":\"\"},\"index\":0,\"finish_reason\":\"stop\",\"logprobs\":null}],\"created\":1788836967,\"object\":\"chat.completion.chunk\",\"usage\":null}\n\ndata: {\"choices\":[],\"created\":1788836967,\"id\":\"chatcmpl-d8fbc8fe-4d5a-9dc9-84c6-ed0c7b9dd3bb\",\"model\":\"qwen3.8-max\",\"object\":\"chat.completion.chunk\",\"usage\":{\"completion_tokens\":5,\"prompt_tokens\":32,\"prompt_tokens_details\":{\"cached_tokens\":0,\"text_tokens\":32},\"total_tokens\":37}}\n\ndata: [DONE]\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
@@ -1,35 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-messages",
"provider:alibaba",
"protocol:alibaba-messages",
"region:ap-southeast-1",
"thinking",
"usage"
],
"name": "alibaba-messages/qwen-3-7-plus-streams-thinking-disabled",
"recordedAt": "2026-09-08T03:10:57.819Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.7-plus\",\"messages\":[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What is 173 multiplied by 219? Reply with only the final integer.\"}]}],\"stream\":true,\"max_tokens\":4096,\"thinking\":{\"type\":\"disabled\"}}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream"
},
"body": "event:ping\ndata:{\"type\":\"ping\"}\n\nevent:message_start\ndata:{\"message\":{\"model\":\"qwen3.7-plus\",\"id\":\"msg_c4d58b4f-a61d-9d0c-a9e4-cb45d34b1120\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"usage\":{\"input_tokens\":20,\"output_tokens\":0}},\"type\":\"message_start\"}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"type\":\"text\",\"text\":\"\"},\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"3\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"7887\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":0}\n\nevent:message_delta\ndata:{\"delta\":{\"stop_reason\":\"end_turn\"},\"type\":\"message_delta\",\"usage\":{\"output_tokens\":5,\"cache_creation_input_tokens\":0,\"input_tokens\":32,\"cache_read_input_tokens\":0,\"prompt_tokens_details\":{\"cached_tokens\":0}}}\n\nevent:message_stop\ndata:{\"type\":\"message_stop\"}\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,34 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-messages",
"provider:alibaba",
"protocol:alibaba-messages",
"region:ap-southeast-1",
"structured-output"
],
"name": "alibaba-messages/qwen-3-8-max-follows-a-json-schema",
"recordedAt": "2026-09-08T03:11:30.530Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Return a JSON object with one key \\\"city\\\" set to the capital city of France.\"}]}],\"stream\":true,\"max_tokens\":1024,\"thinking\":{\"type\":\"disabled\"},\"output_config\":{\"format\":{\"type\":\"json_schema\",\"schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\"}},\"required\":[\"city\"],\"additionalProperties\":false}}}}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream"
},
"body": "event:ping\ndata:{\"type\":\"ping\"}\n\nevent:message_start\ndata:{\"message\":{\"model\":\"qwen3.8-max\",\"id\":\"msg_16e067c1-984b-94d3-9abb-11792d794271\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"usage\":{\"input_tokens\":18,\"output_tokens\":0}},\"type\":\"message_start\"}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"type\":\"text\",\"text\":\"\"},\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"{\\\"\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"city\\\":\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\" \\\"Paris\\\"}\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":0}\n\nevent:message_delta\ndata:{\"delta\":{\"stop_reason\":\"end_turn\"},\"type\":\"message_delta\",\"usage\":{\"output_tokens\":6,\"cache_creation_input_tokens\":0,\"input_tokens\":32,\"cache_read_input_tokens\":0,\"prompt_tokens_details\":{\"cached_tokens\":0}}}\n\nevent:message_stop\ndata:{\"type\":\"message_stop\"}\n\n"
}
}
]
}
@@ -1,35 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-messages",
"provider:alibaba",
"protocol:alibaba-messages",
"region:ap-southeast-1",
"tool",
"tool-choice"
],
"name": "alibaba-messages/qwen-3-8-max-obeys-named-tool-choice",
"recordedAt": "2026-09-08T03:11:06.590Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"Find the current weather in Paris.\"}]}],\"tools\":[{\"name\":\"get_weather\",\"description\":\"Get weather in a city\",\"input_schema\":{\"type\":\"object\",\"properties\":{\"city\":{\"type\":\"string\",\"enum\":[\"Paris\"]}},\"required\":[\"city\"]}}],\"tool_choice\":{\"type\":\"tool\",\"name\":\"get_weather\"},\"stream\":true,\"max_tokens\":4096,\"thinking\":{\"type\":\"disabled\"}}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream"
},
"body": "event:ping\ndata:{\"type\":\"ping\"}\n\nevent:message_start\ndata:{\"message\":{\"model\":\"qwen3.8-max\",\"id\":\"msg_0e8f1abf-f2bc-9a86-a6a7-14804ac17eac\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"usage\":{\"input_tokens\":45,\"output_tokens\":0}},\"type\":\"message_start\"}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"name\":\"get_weather\",\"input\":{},\"id\":\"toolu_cf9cab33261f4709ae096d8a\",\"type\":\"tool_use\"},\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"partial_json\":\"\",\"type\":\"input_json_delta\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"partial_json\":\"{\\\"city\\\": \\\"Paris\",\"type\":\"input_json_delta\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"partial_json\":\"\\\"\",\"type\":\"input_json_delta\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"partial_json\":\"}\",\"type\":\"input_json_delta\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":0}\n\nevent:message_delta\ndata:{\"delta\":{\"stop_reason\":\"end_turn\"},\"type\":\"message_delta\",\"usage\":{\"output_tokens\":19,\"cache_creation_input_tokens\":0,\"input_tokens\":288,\"cache_read_input_tokens\":0,\"prompt_tokens_details\":{\"cached_tokens\":0}}}\n\nevent:message_stop\ndata:{\"type\":\"message_stop\"}\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,36 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-messages",
"provider:alibaba",
"protocol:alibaba-messages",
"region:ap-southeast-1",
"text",
"reasoning",
"usage"
],
"name": "alibaba-messages/qwen-3-8-max-streams-max-effort",
"recordedAt": "2026-09-08T03:12:36.479Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What is 173 multiplied by 219? Reply with only the final integer.\"}]}],\"stream\":true,\"max_tokens\":4096,\"output_config\":{\"effort\":\"max\"}}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream"
},
"body": "event:ping\ndata:{\"type\":\"ping\"}\n\nevent:message_start\ndata:{\"message\":{\"model\":\"qwen3.8-max\",\"id\":\"msg_7952446b-0176-9526-ad7d-b1f0184bc106\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"usage\":{\"input_tokens\":20,\"output_tokens\":0}},\"type\":\"message_start\"}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"type\":\"thinking\",\"signature\":\"\",\"thinking\":\"\"},\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"We\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" need answer user\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"'s simple multiplication with\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" only final integer.\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" Need compute 1\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"73*2\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"19. \"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"173*\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"200=\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"3460\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"0; 1\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"73*1\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"9=32\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"87 (\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"173*\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"20=3\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"460-\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"173=\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"3287\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"); sum=3\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"7887\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\". Final only integer\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\".\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"signature_delta\",\"signature\":\"\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":0}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"type\":\"text\",\"text\":\"\"},\"index\":1}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"37\"},\"type\":\"content_block_delta\",\"index\":1}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"887\"},\"type\":\"content_block_delta\",\"index\":1}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":1}\n\nevent:message_delta\ndata:{\"delta\":{\"stop_reason\":\"end_turn\"},\"type\":\"message_delta\",\"usage\":{\"output_tokens\":92,\"cache_creation_input_tokens\":0,\"input_tokens\":81,\"cache_read_input_tokens\":0,\"prompt_tokens_details\":{\"cached_tokens\":0}}}\n\nevent:message_stop\ndata:{\"type\":\"message_stop\"}\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
@@ -1,36 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-messages",
"provider:alibaba",
"protocol:alibaba-messages",
"region:ap-southeast-1",
"text",
"reasoning",
"usage"
],
"name": "alibaba-messages/qwen-3-8-max-streams-xhigh-effort",
"recordedAt": "2026-09-08T03:10:04.633Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/apps/anthropic/v1/messages",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"messages\":[{\"role\":\"user\",\"content\":[{\"type\":\"text\",\"text\":\"What is 173 multiplied by 219? Reply with only the final integer.\"}]}],\"stream\":true,\"max_tokens\":4096,\"output_config\":{\"effort\":\"xhigh\"}}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream"
},
"body": "event:ping\ndata:{\"type\":\"ping\"}\n\nevent:message_start\ndata:{\"message\":{\"model\":\"qwen3.8-max\",\"id\":\"msg_a57b8563-71fb-9248-a31e-c49d6ba38717\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"usage\":{\"input_tokens\":20,\"output_tokens\":0}},\"type\":\"message_start\"}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"type\":\"thinking\",\"signature\":\"\",\"thinking\":\"\"},\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"We\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" need answer simple multiplication\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\". We\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" already call\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\". Need compute \"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"173*\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"219.\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" 173\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"*200\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"=346\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"00; *\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"19=3\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"287;\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" sum 37\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"88\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\"7. Final only\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" integer. Ensure\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"thinking_delta\",\"thinking\":\" no extra.\\n\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"signature_delta\",\"signature\":\"\"},\"type\":\"content_block_delta\",\"index\":0}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":0}\n\nevent:content_block_start\ndata:{\"type\":\"content_block_start\",\"content_block\":{\"type\":\"text\",\"text\":\"\"},\"index\":1}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"378\"},\"type\":\"content_block_delta\",\"index\":1}\n\nevent:content_block_delta\ndata:{\"delta\":{\"type\":\"text_delta\",\"text\":\"87\"},\"type\":\"content_block_delta\",\"index\":1}\n\nevent:content_block_stop\ndata:{\"type\":\"content_block_stop\",\"index\":1}\n\nevent:message_delta\ndata:{\"delta\":{\"stop_reason\":\"end_turn\"},\"type\":\"message_delta\",\"usage\":{\"output_tokens\":69,\"cache_creation_input_tokens\":0,\"input_tokens\":81,\"cache_read_input_tokens\":0,\"prompt_tokens_details\":{\"cached_tokens\":0}}}\n\nevent:message_stop\ndata:{\"type\":\"message_stop\"}\n\n"
}
}
]
}
@@ -1,35 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-responses",
"provider:alibaba",
"protocol:alibaba-responses",
"region:ap-southeast-1",
"thinking",
"usage"
],
"name": "alibaba-responses/qwen-3-7-plus-streams-thinking-disabled",
"recordedAt": "2026-09-08T03:11:07.495Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/responses",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.7-plus\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"What is 173 multiplied by 219? Reply with only the final integer.\"}]}],\"max_output_tokens\":4096,\"enable_thinking\":false,\"stream\":true}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=UTF-8"
},
"body": "id:1\nevent:response.created\n:HTTP_STATUS/200\ndata:{\"sequence_number\":0,\"type\":\"response.created\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"created_at\":1788837067,\"store\":true,\"tools\":[],\"output\":[],\"top_p\":1.0,\"completed_at\":1788837067,\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.7-plus\",\"service_tier\":\"default\",\"id\":\"resp_493607bf-baef-9ced-9e36-2e2a583b6177\",\"max_output_tokens\":4096,\"object\":\"response\",\"status\":\"queued\"}}\n\nid:2\nevent:response.in_progress\n:HTTP_STATUS/200\ndata:{\"sequence_number\":1,\"type\":\"response.in_progress\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"created_at\":1788837067,\"store\":true,\"tools\":[],\"output\":[],\"top_p\":1.0,\"completed_at\":1788837067,\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.7-plus\",\"service_tier\":\"default\",\"id\":\"resp_493607bf-baef-9ced-9e36-2e2a583b6177\",\"max_output_tokens\":4096,\"object\":\"response\",\"status\":\"in_progress\"}}\n\nid:3\nevent:response.output_item.added\n:HTTP_STATUS/200\ndata:{\"sequence_number\":2,\"item\":{\"id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"status\":\"in_progress\"},\"output_index\":0,\"type\":\"response.output_item.added\"}\n\nid:4\nevent:response.content_part.added\n:HTTP_STATUS/200\ndata:{\"sequence_number\":3,\"output_index\":0,\"type\":\"response.content_part.added\",\"content_index\":0,\"item_id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"part\":{\"type\":\"output_text\",\"annotations\":[],\"text\":\"\"}}\n\nid:5\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":4,\"content_index\":0,\"item_id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"delta\":\"3\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:6\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":5,\"content_index\":0,\"item_id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"delta\":\"7887\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:7\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":6,\"content_index\":0,\"item_id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"delta\":\"\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:8\nevent:response.output_text.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":7,\"content_index\":0,\"item_id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"text\":\"37887\",\"output_index\":0,\"type\":\"response.output_text.done\",\"logprobs\":[]}\n\nid:9\nevent:response.content_part.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":8,\"output_index\":0,\"type\":\"response.content_part.done\",\"content_index\":0,\"item_id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"part\":{\"type\":\"output_text\",\"annotations\":[],\"text\":\"37887\"}}\n\nid:10\nevent:response.output_item.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":9,\"item\":{\"id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[{\"type\":\"output_text\",\"annotations\":[],\"text\":\"37887\"}],\"status\":\"completed\"},\"output_index\":0,\"type\":\"response.output_item.done\"}\n\nid:11\nevent:response.completed\n:HTTP_STATUS/200\ndata:{\"sequence_number\":10,\"type\":\"response.completed\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"usage\":{\"total_tokens\":73,\"input_tokens_details\":{\"cached_tokens\":0},\"output_tokens\":5,\"input_tokens\":68,\"output_tokens_details\":{\"reasoning_tokens\":0},\"x_details\":[{\"total_tokens\":73,\"x_billing_type\":\"response_api\",\"output_tokens\":5,\"input_tokens\":68,\"prompt_tokens_details\":{\"cached_tokens\":0}}]},\"created_at\":1788837067,\"store\":true,\"tools\":[],\"output\":[{\"id\":\"msg_509c6dec-e527-41b5-a9c9-7f02fecb7a3f\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[{\"type\":\"output_text\",\"annotations\":[],\"text\":\"37887\"}],\"status\":\"completed\"}],\"top_p\":1.0,\"completed_at\":1788837067,\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.7-plus\",\"service_tier\":\"default\",\"id\":\"resp_493607bf-baef-9ced-9e36-2e2a583b6177\",\"max_output_tokens\":4096,\"object\":\"response\",\"status\":\"completed\"}}\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,53 +0,0 @@
{
"version": 1,
"metadata": {
"tags": [
"prefix:alibaba-responses",
"provider:alibaba",
"protocol:alibaba-responses",
"region:ap-southeast-1",
"continuation",
"storage"
],
"name": "alibaba-responses/qwen-3-8-max-continues-a-stored-response",
"recordedAt": "2026-09-08T03:12:42.429Z"
},
"interactions": [
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/responses",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Remember the password word apricot. Reply OK.\"}]}],\"store\":true,\"reasoning\":{\"effort\":\"none\"},\"max_output_tokens\":1024,\"stream\":true}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=UTF-8"
},
"body": "id:1\nevent:response.created\n:HTTP_STATUS/200\ndata:{\"sequence_number\":0,\"type\":\"response.created\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"reasoning\":{\"effort\":\"none\"},\"created_at\":1788837161,\"store\":true,\"tools\":[],\"output\":[],\"top_p\":1.0,\"completed_at\":1788837161,\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.8-max\",\"service_tier\":\"default\",\"id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"max_output_tokens\":1024,\"object\":\"response\",\"status\":\"queued\"}}\n\nid:2\nevent:response.in_progress\n:HTTP_STATUS/200\ndata:{\"sequence_number\":1,\"type\":\"response.in_progress\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"reasoning\":{\"effort\":\"none\"},\"created_at\":1788837161,\"store\":true,\"tools\":[],\"output\":[],\"top_p\":1.0,\"completed_at\":1788837161,\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.8-max\",\"service_tier\":\"default\",\"id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"max_output_tokens\":1024,\"object\":\"response\",\"status\":\"in_progress\"}}\n\nid:3\nevent:response.output_item.added\n:HTTP_STATUS/200\ndata:{\"sequence_number\":2,\"item\":{\"id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"status\":\"in_progress\"},\"output_index\":0,\"type\":\"response.output_item.added\"}\n\nid:4\nevent:response.content_part.added\n:HTTP_STATUS/200\ndata:{\"sequence_number\":3,\"output_index\":0,\"type\":\"response.content_part.added\",\"content_index\":0,\"item_id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"part\":{\"type\":\"output_text\",\"annotations\":[],\"text\":\"\"}}\n\nid:5\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":4,\"content_index\":0,\"item_id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"delta\":\"OK\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:6\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":5,\"content_index\":0,\"item_id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"delta\":\".\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:7\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":6,\"content_index\":0,\"item_id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"delta\":\"\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:8\nevent:response.output_text.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":7,\"content_index\":0,\"item_id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"text\":\"OK.\",\"output_index\":0,\"type\":\"response.output_text.done\",\"logprobs\":[]}\n\nid:9\nevent:response.content_part.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":8,\"output_index\":0,\"type\":\"response.content_part.done\",\"content_index\":0,\"item_id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"part\":{\"type\":\"output_text\",\"annotations\":[],\"text\":\"OK.\"}}\n\nid:10\nevent:response.output_item.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":9,\"item\":{\"id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[{\"type\":\"output_text\",\"annotations\":[],\"text\":\"OK.\"}],\"status\":\"completed\"},\"output_index\":0,\"type\":\"response.output_item.done\"}\n\nid:11\nevent:response.completed\n:HTTP_STATUS/200\ndata:{\"sequence_number\":10,\"type\":\"response.completed\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"reasoning\":{\"effort\":\"none\"},\"usage\":{\"total_tokens\":60,\"input_tokens_details\":{\"cached_tokens\":0},\"output_tokens\":2,\"input_tokens\":58,\"output_tokens_details\":{\"reasoning_tokens\":0},\"x_details\":[{\"total_tokens\":60,\"x_billing_type\":\"response_api\",\"output_tokens\":2,\"input_tokens\":58,\"prompt_tokens_details\":{\"cached_tokens\":0}}]},\"created_at\":1788837161,\"store\":true,\"tools\":[],\"output\":[{\"id\":\"msg_a908ad8e-6220-444b-880c-02aecbaf8fac\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[{\"type\":\"output_text\",\"annotations\":[],\"text\":\"OK.\"}],\"status\":\"completed\"}],\"top_p\":1.0,\"completed_at\":1788837161,\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.8-max\",\"service_tier\":\"default\",\"id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"max_output_tokens\":1024,\"object\":\"response\",\"status\":\"completed\"}}\n\n"
}
},
{
"transport": "http",
"request": {
"method": "POST",
"url": "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/responses",
"headers": {
"content-type": "application/json"
},
"body": "{\"model\":\"qwen3.8-max\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"What word did I ask you to remember? Reply with only the word.\"}]}],\"store\":true,\"reasoning\":{\"effort\":\"none\"},\"max_output_tokens\":1024,\"previous_response_id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"stream\":true}"
},
"response": {
"status": 200,
"headers": {
"content-type": "text/event-stream;charset=UTF-8"
},
"body": "id:1\nevent:response.created\n:HTTP_STATUS/200\ndata:{\"sequence_number\":0,\"type\":\"response.created\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"reasoning\":{\"effort\":\"none\"},\"created_at\":1788837162,\"store\":true,\"tools\":[],\"output\":[],\"top_p\":1.0,\"completed_at\":1788837162,\"previous_response_id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.8-max\",\"service_tier\":\"default\",\"id\":\"resp_1d81d34c-d3ec-9f98-ab59-5d4641d4217a\",\"max_output_tokens\":1024,\"object\":\"response\",\"status\":\"queued\"}}\n\nid:2\nevent:response.in_progress\n:HTTP_STATUS/200\ndata:{\"sequence_number\":1,\"type\":\"response.in_progress\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"reasoning\":{\"effort\":\"none\"},\"created_at\":1788837162,\"store\":true,\"tools\":[],\"output\":[],\"top_p\":1.0,\"completed_at\":1788837162,\"previous_response_id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.8-max\",\"service_tier\":\"default\",\"id\":\"resp_1d81d34c-d3ec-9f98-ab59-5d4641d4217a\",\"max_output_tokens\":1024,\"object\":\"response\",\"status\":\"in_progress\"}}\n\nid:3\nevent:response.output_item.added\n:HTTP_STATUS/200\ndata:{\"sequence_number\":2,\"item\":{\"id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[],\"status\":\"in_progress\"},\"output_index\":0,\"type\":\"response.output_item.added\"}\n\nid:4\nevent:response.content_part.added\n:HTTP_STATUS/200\ndata:{\"sequence_number\":3,\"output_index\":0,\"type\":\"response.content_part.added\",\"content_index\":0,\"item_id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"part\":{\"type\":\"output_text\",\"annotations\":[],\"text\":\"\"}}\n\nid:5\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":4,\"content_index\":0,\"item_id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"delta\":\"ap\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:6\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":5,\"content_index\":0,\"item_id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"delta\":\"ricot\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:7\nevent:response.output_text.delta\n:HTTP_STATUS/200\ndata:{\"sequence_number\":6,\"content_index\":0,\"item_id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"delta\":\"\",\"output_index\":0,\"type\":\"response.output_text.delta\",\"logprobs\":[]}\n\nid:8\nevent:response.output_text.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":7,\"content_index\":0,\"item_id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"text\":\"apricot\",\"output_index\":0,\"type\":\"response.output_text.done\",\"logprobs\":[]}\n\nid:9\nevent:response.content_part.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":8,\"output_index\":0,\"type\":\"response.content_part.done\",\"content_index\":0,\"item_id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"part\":{\"type\":\"output_text\",\"annotations\":[],\"text\":\"apricot\"}}\n\nid:10\nevent:response.output_item.done\n:HTTP_STATUS/200\ndata:{\"sequence_number\":9,\"item\":{\"id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[{\"type\":\"output_text\",\"annotations\":[],\"text\":\"apricot\"}],\"status\":\"completed\"},\"output_index\":0,\"type\":\"response.output_item.done\"}\n\nid:11\nevent:response.completed\n:HTTP_STATUS/200\ndata:{\"sequence_number\":10,\"type\":\"response.completed\",\"response\":{\"top_logprobs\":0,\"metadata\":{},\"presence_penalty\":0.0,\"reasoning\":{\"effort\":\"none\"},\"usage\":{\"total_tokens\":92,\"input_tokens_details\":{\"cached_tokens\":0},\"output_tokens\":3,\"input_tokens\":89,\"output_tokens_details\":{\"reasoning_tokens\":0},\"x_details\":[{\"total_tokens\":92,\"x_billing_type\":\"response_api\",\"output_tokens\":3,\"input_tokens\":89,\"prompt_tokens_details\":{\"cached_tokens\":0}}]},\"created_at\":1788837162,\"store\":true,\"tools\":[],\"output\":[{\"id\":\"msg_7bc50c33-8fbd-4b10-91fc-4f2533bb2091\",\"role\":\"assistant\",\"type\":\"message\",\"content\":[{\"type\":\"output_text\",\"annotations\":[],\"text\":\"apricot\"}],\"status\":\"completed\"}],\"top_p\":1.0,\"completed_at\":1788837162,\"previous_response_id\":\"resp_97f8cd28-c7ab-9d51-9c14-4fc531decdc8\",\"frequency_penalty\":0.0,\"parallel_tool_calls\":true,\"background\":false,\"temperature\":1.0,\"tool_choice\":\"auto\",\"model\":\"qwen3.8-max\",\"service_tier\":\"default\",\"id\":\"resp_1d81d34c-d3ec-9f98-ab59-5d4641d4217a\",\"max_output_tokens\":1024,\"object\":\"response\",\"status\":\"completed\"}}\n\n"
}
}
]
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -81,7 +81,7 @@
{
"direction": "client",
"kind": "text",
"body": "{\"type\":\"response.create\",\"model\":\"gpt-5.5\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Alpha.\"}]},{\"type\":\"message\",\"id\":\"msg_ws_reconnect_1\",\"role\":\"assistant\",\"status\":\"completed\",\"content\":[{\"type\":\"output_text\",\"text\":\"Alpha.\"}]},{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Beta.\"}]}],\"store\":false,\"max_output_tokens\":30,\"include\":[\"reasoning.encrypted_content\"],\"reasoning\":{\"effort\":\"medium\",\"summary\":\"auto\"},\"text\":{\"verbosity\":\"low\"},\"instructions\":\"Follow the user's exact reply instruction.\"}"
"body": "{\"type\":\"response.create\",\"model\":\"gpt-5.5\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Alpha.\"}]},{\"type\":\"message\",\"id\":\"msg_ws_reconnect_1\",\"role\":\"assistant\",\"content\":[{\"type\":\"output_text\",\"text\":\"Alpha.\"}]},{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Beta.\"}]}],\"store\":false,\"max_output_tokens\":30,\"include\":[\"reasoning.encrypted_content\"],\"reasoning\":{\"effort\":\"medium\",\"summary\":\"auto\"},\"text\":{\"verbosity\":\"low\"},\"instructions\":\"Follow the user's exact reply instruction.\"}"
},
{
"direction": "server",
@@ -91,7 +91,7 @@
{
"direction": "client",
"kind": "text",
"body": "{\"type\":\"response.create\",\"model\":\"gpt-5.5\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Ready.\"}]},{\"type\":\"message\",\"id\":\"msg_ws_rejection_1\",\"role\":\"assistant\",\"status\":\"completed\",\"content\":[{\"type\":\"output_text\",\"text\":\"Ready.\"}]},{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Recovered.\"}]}],\"store\":false,\"max_output_tokens\":30,\"include\":[\"reasoning.encrypted_content\"],\"reasoning\":{\"effort\":\"medium\",\"summary\":\"auto\"},\"text\":{\"verbosity\":\"low\"},\"instructions\":\"Follow the user's exact reply instruction.\"}"
"body": "{\"type\":\"response.create\",\"model\":\"gpt-5.5\",\"input\":[{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Ready.\"}]},{\"type\":\"message\",\"id\":\"msg_ws_rejection_1\",\"role\":\"assistant\",\"content\":[{\"type\":\"output_text\",\"text\":\"Ready.\"}]},{\"role\":\"user\",\"content\":[{\"type\":\"input_text\",\"text\":\"Reply exactly: Recovered.\"}]}],\"store\":false,\"max_output_tokens\":30,\"include\":[\"reasoning.encrypted_content\"],\"reasoning\":{\"effort\":\"medium\",\"summary\":\"auto\"},\"text\":{\"verbosity\":\"low\"},\"instructions\":\"Follow the user's exact reply instruction.\"}"
},
{
"direction": "server",
File diff suppressed because one or more lines are too long
@@ -1,39 +0,0 @@
import { LLM } from "../../src/index.js"
import { Alibaba } from "../../src/providers.js"
const provider = Alibaba.configure({ region: "ap-southeast-1" })
Alibaba.configure({ baseURL: "https://gateway.example/v1" })
Alibaba.configure({ region: "eu-central-1", workspaceID: "llm-workspace" })
LLM.request({
model: provider.chat("qwen3.8-max"),
providerOptions: { reasoningEffort: "future", enableThinking: true, preserveThinking: false, toolStream: true },
})
LLM.request({
model: provider.messages("qwen3.8-max"),
providerOptions: { effort: "xhigh", thinking: { type: "enabled" } },
})
LLM.request({
model: provider.messages("qwen3.7-plus"),
providerOptions: { thinking: { type: "enabled", budgetTokens: 512 } },
})
LLM.request({
model: provider.responses("qwen3.8-max"),
providerOptions: { reasoningEffort: "low", enableThinking: true, store: true, previousResponseId: "resp_previous" },
})
// @ts-expect-error Region or complete base URL is required.
Alibaba.configure({ apiKey: "fixture" })
LLM.request({
model: provider.chat("qwen3.8-max"),
// @ts-expect-error Thinking toggle is a boolean.
providerOptions: { enableThinking: "true" },
})
LLM.request({
model: provider.messages("qwen3.8-max"),
// @ts-expect-error Messages uses effort.
providerOptions: { reasoningEffort: "high" },
})
LLM.request({
model: provider.responses("qwen3.8-max"),
// @ts-expect-error Responses does not use Chat thinking budgets.
providerOptions: { thinkingBudget: 512 },
})
+11 -48
View File
@@ -3,9 +3,6 @@ import { model } from "@opencode/ai/providers/openai"
import { LLM } from "../src/index.js"
import { Endpoint } from "../src/route/endpoint.js"
const configuration = (provider: string, message: string) =>
expect.objectContaining({ _tag: "ProviderConfiguration", provider, message })
describe("provider package entrypoints", () => {
test("semantic API aliases expose the same contract", async () => {
const modules = await Promise.all([
@@ -54,10 +51,6 @@ describe("provider package entrypoints", () => {
import("@opencode/ai/providers/zai-coding-plan/chat"),
import("@opencode/ai/providers/zai-coding-plan/messages"),
import("@opencode/ai/providers/zai-coding-plan/responses"),
import("@opencode/ai/providers/alibaba"),
import("@opencode/ai/providers/alibaba/chat"),
import("@opencode/ai/providers/alibaba/messages"),
import("@opencode/ai/providers/alibaba/responses"),
])
for (const module of modules) expect(module.model).toBeFunction()
@@ -68,34 +61,6 @@ describe("provider package entrypoints", () => {
expect(modules[19].model).not.toBe(modules[20].model)
})
test("maps Alibaba API entrypoints onto explicit regional routes", async () => {
const modules = await Promise.all([
import("@opencode/ai/providers/alibaba"),
import("@opencode/ai/providers/alibaba/chat"),
import("@opencode/ai/providers/alibaba/messages"),
import("@opencode/ai/providers/alibaba/responses"),
])
expect(modules[0].model).toBe(modules[1].model)
const settings = {
region: "eu-central-1",
workspaceID: "llm-fixture",
apiKey: "fixture",
headers: { "x-test": "fixture" },
body: { extension: true },
}
const routes = ["alibaba-chat", "alibaba-chat", "alibaba-messages", "alibaba-responses"]
modules.forEach((module, index) => {
const model = module.model("qwen3.8-max", settings)
expect(model.provider).toBe("alibaba")
expect(model.route.id).toBe(routes[index])
expect(model.route.endpoint.baseURL).toBe(
`https://llm-fixture.eu-central-1.maas.aliyuncs.com/${index === 2 ? "apps/anthropic/v1" : "compatible-mode/v1"}`,
)
expect(model.route.defaults.headers).toEqual(settings.headers)
expect(model.route.defaults.http?.body).toEqual(settings.body)
})
})
test("maps Moonshot API entrypoints onto provider-owned routes", async () => {
const modules = await Promise.all([
import("@opencode/ai/providers/moonshot"),
@@ -325,7 +290,7 @@ describe("provider package entrypoints", () => {
const AnthropicCompatible = await import("@opencode/ai/providers/anthropic-compatible")
expect(() =>
Reflect.apply(AnthropicCompatible.model, undefined, ["compatible-model", { apiKey: "fixture" }]),
).toThrow(configuration("anthropic-compatible", "Anthropic-compatible providers require a baseURL"))
).toThrow("Anthropic-compatible providers require a baseURL")
})
test("rejects conflicting Anthropic-compatible auth settings at runtime", async () => {
@@ -340,10 +305,10 @@ describe("provider package entrypoints", () => {
baseURL: "https://messages.example.test/v1",
},
]),
).toThrow(configuration("anthropic-compatible", "Anthropic-compatible apiKey cannot be combined with authToken"))
).toThrow("Anthropic-compatible apiKey cannot be combined with authToken")
expect(() =>
Reflect.apply(Anthropic.model, undefined, ["claude-sonnet-4-6", { apiKey: "fixture", authToken: "token" }]),
).toThrow(configuration("anthropic", "Anthropic apiKey cannot be combined with authToken"))
).toThrow("Anthropic apiKey cannot be combined with authToken")
})
test("maps legacy OpenAI organization and project settings to headers", () => {
@@ -493,45 +458,43 @@ describe("provider package entrypoints", () => {
"gemini-3.5-flash",
{ accessToken: "token", apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex apiKey cannot be combined with accessToken or auth"))
).toThrow("Google Vertex apiKey cannot be combined with accessToken or auth")
const configured = Reflect.apply(GoogleVertex.configure, undefined, [
{ accessToken: "token", auth: {}, project: "vertex-project" },
])
expect(() => configured.model("gemini-3.5-flash")).toThrow(
configuration("google-vertex", "Google Vertex accessToken cannot be combined with auth"),
)
expect(() => configured.model("gemini-3.5-flash")).toThrow("Google Vertex accessToken cannot be combined with auth")
expect(() =>
Reflect.apply(GoogleVertexMessages.model, undefined, [
"claude-sonnet-4-6",
{ apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex Messages does not support API keys"))
).toThrow("Google Vertex Messages does not support API keys")
expect(() =>
Reflect.apply(Providers.GoogleVertexMessages.configure, undefined, [
{ apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex Messages does not support API keys"))
).toThrow("Google Vertex Messages does not support API keys")
expect(() =>
Reflect.apply(GoogleVertexChat.model, undefined, [
"deepseek-ai/deepseek-v3.2-maas",
{ apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex Chat does not support API keys"))
).toThrow("Google Vertex Chat does not support API keys")
expect(() =>
Reflect.apply(Providers.GoogleVertexChat.configure, undefined, [
{ apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex Chat does not support API keys"))
).toThrow("Google Vertex Chat does not support API keys")
expect(() =>
Reflect.apply(GoogleVertexResponses.model, undefined, [
"xai/grok-4.20-reasoning",
{ apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex Responses does not support API keys"))
).toThrow("Google Vertex Responses does not support API keys")
expect(() =>
Reflect.apply(Providers.GoogleVertexResponses.configure, undefined, [
{ apiKey: "fixture", project: "vertex-project" },
]),
).toThrow(configuration("google-vertex", "Google Vertex Responses does not support API keys"))
).toThrow("Google Vertex Responses does not support API keys")
})
})
@@ -1,254 +0,0 @@
import { describe, expect } from "bun:test"
import { Effect } from "effect"
import { LLM, LLMEvent, LLMRequest, Message, ToolDefinition } from "../../src/index.js"
import { Alibaba } from "../../src/providers.js"
import { LLMClient } from "../../src/route.js"
import { compileRequest } from "../../src/route/client.js"
import { recordedTests } from "../recorded-test.js"
const alibaba = Alibaba.configure({ region: "ap-southeast-1", apiKey: process.env.ALIBABA_API_KEY ?? "fixture" })
const record = (api: "chat" | "messages" | "responses") =>
recordedTests({
prefix: `alibaba-${api}`,
provider: "alibaba",
protocol: `alibaba-${api}`,
requires: ["ALIBABA_API_KEY"],
tags: ["region:ap-southeast-1"],
})
for (const api of ["chat", "messages", "responses"] as const) {
const recorded = record(api)
describe(`Alibaba ${api} capabilities`, () => {
for (const enabled of [false, true]) {
recorded.effect.with(
`Qwen 3.7 Plus streams thinking ${enabled ? "enabled" : "disabled"}`,
{ tags: ["thinking", "usage"] },
() =>
Effect.gen(function* () {
const request = LLM.request({
model: alibaba[api]("qwen3.7-plus"),
providerOptions:
api === "messages"
? { thinking: { type: enabled ? "enabled" : "disabled", ...(enabled ? { budgetTokens: 1024 } : {}) } }
: api === "chat"
? { enableThinking: enabled, ...(enabled ? { thinkingBudget: 1024 } : {}) }
: { enableThinking: enabled },
prompt: "What is 173 multiplied by 219? Reply with only the final integer.",
generation: { maxTokens: 4096 },
})
const compiled = yield* compileRequest(request)
expect(api === "messages" ? compiled.body.thinking.type : compiled.body.enable_thinking).toBe(
api === "messages" ? (enabled ? "enabled" : "disabled") : enabled,
)
const response = yield* LLMClient.generate(request)
expect(response.text.replaceAll(",", "")).toContain("37887")
expect(response.reasoning.length > 0).toBe(enabled)
expect(response.finishReason.normalized).toBe("stop")
expect(response.usage?.inputTokens).toBeGreaterThan(0)
expect(response.usage?.outputTokens).toBeGreaterThan(0)
}),
120_000,
)
}
recorded.effect.with(
"Qwen 3.8 Flash reads image bytes",
{ tags: ["image"] },
() =>
Effect.gen(function* () {
const bytes = yield* Effect.promise(() =>
Bun.file(new URL("../fixtures/media/restroom.png", import.meta.url)).bytes(),
)
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba[api]("qwen3.8-flash"),
providerOptions: api === "messages" ? { thinking: { type: "disabled" } } : { enableThinking: false },
messages: [
Message.user([
{ type: "text", text: "Read the three words in this image. Reply with only the words in order." },
{ type: "media", mediaType: "image/png", data: bytes },
]),
],
generation: { maxTokens: 4096 },
}),
)
expect(response.text.toLowerCase()).toContain("jiggling restroom prison")
expect(response.finishReason.normalized).toBe("stop")
}),
120_000,
)
recorded.effect.with(
"Qwen 3.8 Max obeys named tool choice",
{ tags: ["tool", "tool-choice"] },
() =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba[api]("qwen3.8-max"),
prompt: "Find the current weather in Paris.",
providerOptions: api === "messages" ? { thinking: { type: "disabled" } } : { reasoningEffort: "none" },
tools: [
ToolDefinition.make({
name: "get_weather",
description: "Get weather in a city",
inputSchema: {
type: "object",
properties: { city: { type: "string", enum: ["Paris"] } },
required: ["city"],
},
}),
],
toolChoice: { type: "tool", name: "get_weather" },
generation: { maxTokens: 4096 },
}),
)
expect(response.toolCalls).toMatchObject([{ name: "get_weather", input: { city: "Paris" } }])
expect(response.finishReason.normalized).toBe(api === "messages" ? "stop" : "tool-calls")
if (api === "messages") expect(response.finishReason.raw).toBe("end_turn")
}),
120_000,
)
})
}
record("chat").effect.with(
"Qwen 3.8 Max returns a JSON object",
{ tags: ["structured-output"] },
() =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba.chat("qwen3.8-max"),
prompt: 'Return a JSON object with one key "city" set to the capital city of France.',
providerOptions: { reasoningEffort: "none", responseFormat: { type: "json_object" } },
generation: { maxTokens: 1024 },
}),
)
expect(JSON.parse(response.text)).toEqual({ city: "Paris" })
expect(response.finishReason.normalized).toBe("stop")
}),
120_000,
)
record("messages").effect.with(
"Qwen 3.8 Max follows a JSON schema",
{ tags: ["structured-output"] },
() =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba.messages("qwen3.8-max"),
prompt: 'Return a JSON object with one key "city" set to the capital city of France.',
providerOptions: {
thinking: { type: "disabled" },
outputConfig: {
format: {
type: "json_schema",
schema: {
type: "object",
properties: { city: { type: "string" } },
required: ["city"],
additionalProperties: false,
},
},
},
},
generation: { maxTokens: 1024 },
}),
)
expect(JSON.parse(response.text)).toEqual({ city: "Paris" })
expect(response.finishReason.normalized).toBe("stop")
}),
120_000,
)
const responses = record("responses")
responses.effect.with(
"Qwen 3.8 Max continues a stored response",
{ tags: ["continuation", "storage"] },
() =>
Effect.gen(function* () {
const first = yield* LLMClient.generate(
LLM.request({
model: alibaba.responses("qwen3.8-max"),
prompt: "Remember the password word apricot. Reply OK.",
providerOptions: { store: true, reasoningEffort: "none" },
generation: { maxTokens: 1024 },
}),
)
const id = first.events.find(LLMEvent.is.finish)?.providerMetadata?.alibaba?.responseId
expect(id).toBeString()
if (typeof id !== "string") throw new Error("Missing Alibaba response ID")
const second = yield* LLMClient.generate(
LLM.request({
model: alibaba.responses("qwen3.8-max"),
prompt: "What word did I ask you to remember? Reply with only the word.",
providerOptions: { previousResponseId: id, store: true, reasoningEffort: "none" },
generation: { maxTokens: 1024 },
}),
)
expect(second.text.toLowerCase()).toContain("apricot")
expect(second.finishReason.normalized).toBe("stop")
}),
120_000,
)
responses.effect.with(
"Qwen 3.8 Max uses hosted web search and extraction",
{ tags: ["hosted-tool", "web-search", "web-extractor"] },
() =>
Effect.gen(function* () {
const request = LLM.request({
model: alibaba.responses("qwen3.8-max"),
prompt:
"Use web search to find Alibaba Cloud Model Studio's official documentation, then use web_extractor to read the page. Give a brief summary with the source URL.",
tools: [Alibaba.webSearch(), Alibaba.webExtractor()],
providerOptions: { reasoningEffort: "low", store: false },
generation: { maxTokens: 4096 },
})
const response = yield* LLMClient.generate(request)
expect(response.toolCalls).toEqual(
expect.arrayContaining([
expect.objectContaining({ name: "web_search", providerExecuted: true }),
expect.objectContaining({ name: "web_extractor", providerExecuted: true }),
]),
)
expect(response.text.toLowerCase()).toContain("alibaba")
expect(response.events.some(LLMEvent.is.toolResult)).toBe(true)
const replay = yield* compileRequest(
LLMRequest.update(request, {
messages: [...request.messages, response.message, Message.user("Summarize in one sentence.")],
}),
)
expect(replay.body.input).toEqual(
expect.arrayContaining([
expect.objectContaining({ type: "web_search_call" }),
expect.objectContaining({ type: "web_extractor_call" }),
]),
)
}),
180_000,
)
responses.effect.with(
"Qwen 3.8 Max uses hosted code interpreter",
{ tags: ["hosted-tool", "code-interpreter"] },
() =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(
LLM.request({
model: alibaba.responses("qwen3.8-max"),
prompt:
"Use the code interpreter to compute the SHA-256 hash of the UTF-8 string hello (no newline). Reply with only the hash.",
tools: [Alibaba.codeInterpreter()],
providerOptions: { reasoningEffort: "low" },
generation: { maxTokens: 4096 },
}),
)
expect(response.toolCalls).toEqual(
expect.arrayContaining([expect.objectContaining({ name: "code_interpreter", providerExecuted: true })]),
)
expect(response.text).toContain("2cf24dba5fb0a30e26e83b2ac5b9e29e1b161e5c1fa7425e73043362938b9824")
expect(response.events.some(LLMEvent.is.toolResult)).toBe(true)
}),
180_000,
)
@@ -1,163 +0,0 @@
import { describe, expect } from "bun:test"
import { Effect } from "effect"
import { LLM, LLMEvent, LLMRequest, Message, ToolDefinition, type LLMResponse } from "../../src/index.js"
import { Alibaba } from "../../src/providers.js"
import { LLMClient } from "../../src/route.js"
import { compileRequest } from "../../src/route/client.js"
import { recordedTests } from "../recorded-test.js"
const alibaba = Alibaba.configure({ region: "ap-southeast-1", apiKey: process.env.ALIBABA_API_KEY ?? "fixture" })
const weather = ToolDefinition.make({
name: "get_weather",
description: "Get the current weather in a city",
inputSchema: {
type: "object",
properties: { city: { type: "string", enum: ["Paris"] } },
required: ["city"],
additionalProperties: false,
},
})
for (const api of ["chat", "messages", "responses"] as const) {
const recorded = recordedTests({
prefix: `alibaba-${api}`,
provider: "alibaba",
protocol: `alibaba-${api}`,
requires: ["ALIBABA_API_KEY"],
tags: ["region:ap-southeast-1"],
})
describe(`Alibaba ${api}`, () => {
for (const effort of api === "messages"
? [undefined, "low", "medium", "high", "xhigh", "max"]
: [undefined, "none", "minimal", "low", "medium", "high", "xhigh", "max"]) {
recorded.effect.with(
`Qwen 3.8 Max streams ${effort ?? "default"} effort`,
{ tags: ["text", "reasoning", "usage"] },
() =>
Effect.gen(function* () {
const request = LLM.request({
model: alibaba[api]("qwen3.8-max"),
prompt: "What is 173 multiplied by 219? Reply with only the final integer.",
providerOptions: api === "messages" ? { effort } : { reasoningEffort: effort },
generation: { maxTokens: 4096 },
})
const compiled = yield* compileRequest(request)
expect(compiled.body.enable_thinking).toBeUndefined()
expect(compiled.body.thinking).toBeUndefined()
expect(
api === "chat"
? compiled.body.reasoning_effort
: api === "messages"
? compiled.body.output_config?.effort
: compiled.body.reasoning?.effort,
).toBe(effort)
const response = yield* LLMClient.generate(request)
expect(response.text.replaceAll(",", "")).toContain("37887")
expect(response.reasoning.length > 0).toBe(effort !== "none")
expect(response.finishReason.normalized).toBe("stop")
expectUsage(response)
}),
120_000,
)
}
recorded.effect.with(
"Qwen 3.8 Max replays reasoning through a tool loop and follow-up",
{ tags: ["tool", "tool-loop", "reasoning", "continuation"] },
() =>
Effect.gen(function* () {
const request = LLM.request({
model: alibaba[api]("qwen3.8-max"),
providerOptions:
api === "messages"
? { effort: "medium" }
: api === "chat"
? { reasoningEffort: "medium", preserveThinking: true, toolStream: true }
: { reasoningEffort: "medium", store: false },
prompt:
"We have a budget of 38000 dollars for 219 trips costing 173 dollars each. Calculate whether that is affordable. If it is, use get_weather to look up the current weather in Paris. After receiving the result, report the weather in one short sentence.",
tools: [weather],
generation: { maxTokens: 4096 },
})
const first = yield* LLMClient.generate(request)
expect(first.toolCalls).toMatchObject([{ name: "get_weather", input: { city: "Paris" } }])
expect(first.finishReason.normalized).toBe("tool-calls")
expect(first.reasoning.length).toBeGreaterThan(0)
expect(first.events.some(LLMEvent.is.toolInputDelta)).toBe(true)
expectUsage(first)
const continuation = LLMRequest.update(request, {
messages: [
...request.messages,
first.message,
...first.toolCalls.map((call) =>
Message.tool({ id: call.id, name: call.name, result: { condition: "sunny", temperature: "18C" } }),
),
],
})
expectReasoning(api, (yield* compileRequest(continuation)).body, first)
const second = yield* LLMClient.generate(continuation)
expect(second.text.toLowerCase()).toContain("sunny")
expect(second.toolCalls).toHaveLength(0)
expect(second.finishReason.normalized).toBe("stop")
expectUsage(second)
const followUp = LLMRequest.update(continuation, {
messages: [
...continuation.messages,
second.message,
Message.user("What temperature did the tool report? Reply with only the temperature."),
],
})
expectReasoning(api, (yield* compileRequest(followUp)).body, first)
const third = yield* LLMClient.generate(followUp)
expect(third.text).toContain("18")
expect(third.toolCalls).toHaveLength(0)
expect(third.finishReason.normalized).toBe("stop")
expectUsage(third)
}),
180_000,
)
})
}
function expectUsage(response: LLMResponse) {
expect(response.usage?.inputTokens).toBeGreaterThan(0)
expect(response.usage?.outputTokens).toBeGreaterThan(0)
expect(response.events.filter(LLMEvent.is.finish)).toHaveLength(1)
}
function expectReasoning(
api: "chat" | "messages" | "responses",
body: Readonly<Record<string, unknown>>,
response: LLMResponse,
) {
if (api === "chat") {
expect(body.messages).toEqual(
expect.arrayContaining([expect.objectContaining({ role: "assistant", reasoning_content: response.reasoning })]),
)
return
}
for (const part of response.message.content.filter((part) => part.type === "reasoning")) {
if (api === "messages") {
expect(body.messages).toEqual(
expect.arrayContaining([
expect.objectContaining({
role: "assistant",
content: expect.arrayContaining([
{ type: "thinking", thinking: part.text, signature: part.providerMetadata?.alibaba?.signature ?? "" },
]),
}),
]),
)
continue
}
expect(body.input).toEqual(
expect.arrayContaining([
expect.objectContaining({
type: "reasoning",
id: part.providerMetadata?.alibaba?.itemId,
summary: expect.arrayContaining([{ type: "summary_text", text: part.text }]),
}),
]),
)
}
}
-382
View File
@@ -1,382 +0,0 @@
import { expect, test } from "bun:test"
import { ConfigProvider, Effect } from "effect"
import { Headers } from "effect/unstable/http"
import { Auth, LLM, LLMClient, LLMRequest, Message, ReasoningPart, ToolDefinition } from "../../src/index.js"
import { Alibaba } from "../../src/providers.js"
import { compileRequest } from "../../src/route/client.js"
import { Endpoint } from "../../src/route/endpoint.js"
import { it } from "../lib/effect.js"
import { dynamicResponse } from "../lib/http.js"
import { sseEvents } from "../lib/sse.js"
const tool = ToolDefinition.make({
name: "lookup",
description: "Look up a value",
inputSchema: { type: "object", properties: { key: { type: "string" } } },
})
const paths = {
chat: "/compatible-mode/v1/chat/completions",
messages: "/apps/anthropic/v1/messages",
responses: "/compatible-mode/v1/responses",
}
it.effect("Alibaba owns regional shared and workspace-specific endpoints", () =>
Effect.gen(function* () {
for (const region of [
"ap-southeast-1",
"cn-beijing",
"cn-hongkong",
"us-east-1",
"eu-central-1",
"ap-northeast-1",
]) {
const provider = Alibaba.configure({ region, workspaceID: "llm-fixture", apiKey: "fixture" })
expect(provider.model).toBe(provider.chat)
for (const api of ["chat", "messages", "responses"] as const) {
const model = provider[api]("qwen-plus-us")
const request = LLM.request({ model, prompt: "Hello" })
const compiled = yield* compileRequest(request)
expect(Endpoint.render(model.route.endpoint, { request, body: compiled.body }).toString()).toBe(
`https://llm-fixture.${region}.maas.aliyuncs.com${paths[api]}`,
)
expect(model.provider).toBe("alibaba")
expect(model.route.id).toBe(`alibaba-${api}`)
expect(compiled.body.model).toBe("qwen-plus-us")
for (const field of [
"thinking",
"enable_thinking",
"thinking_budget",
"preserve_thinking",
"reasoning_effort",
"reasoning",
"output_config",
"store",
"tool_stream",
])
expect(compiled.body[field]).toBeUndefined()
}
}
for (const [region, host] of [
["ap-southeast-1", "dashscope-intl.aliyuncs.com"],
["cn-beijing", "dashscope.aliyuncs.com"],
["cn-hongkong", "cn-hongkong.dashscope.aliyuncs.com"],
["us-east-1", "dashscope-us.aliyuncs.com"],
]) {
const provider = Alibaba.configure({ region, apiKey: "fixture" })
for (const api of ["chat", "messages", "responses"] as const) {
const request = LLM.request({ model: provider[api]("qwen3.8-max") })
const compiled = yield* compileRequest(request)
expect(Endpoint.render(request.model.route.endpoint, { request, body: compiled.body }).toString()).toBe(
`https://${host}${paths[api]}`,
)
}
}
}),
)
test("Alibaba requires explicit placement and supports complete base URL overrides", () => {
for (const region of ["eu-central-1", "ap-northeast-1", "future-region"])
expect(() => Alibaba.configure({ region })).toThrow(
expect.objectContaining({
_tag: "ProviderConfiguration",
provider: "alibaba",
message: `Alibaba region ${region} requires workspaceID or baseURL`,
}),
)
for (const config of [
{ baseURL: "https://gateway.example/prefix" },
{ region: "future-region", workspaceID: "ignored", baseURL: "https://gateway.example/prefix" },
]) {
const provider = Alibaba.configure(config)
for (const api of ["chat", "messages", "responses"] as const)
expect(provider[api]("unchanged-id").route.endpoint.baseURL).toBe(config.baseURL)
}
expect(
Alibaba.configure({ region: "future-region", workspaceID: "llm-fixture" }).chat("new-model").route.endpoint.baseURL,
).toBe("https://llm-fixture.future-region.maas.aliyuncs.com/compatible-mode/v1")
})
it.effect("Alibaba resolves explicit auth, API keys, and regional environment credentials", () =>
Effect.gen(function* () {
for (const item of [
{
config: {},
env: { DASHSCOPE_API_KEY: "primary", ALIBABA_API_KEY: "fallback" },
headers: { authorization: "Bearer primary" },
},
{ config: {}, env: { ALIBABA_API_KEY: "fallback" }, headers: { authorization: "Bearer fallback" } },
{
config: { apiKey: "explicit" },
env: { DASHSCOPE_API_KEY: "primary" },
headers: { authorization: "Bearer explicit" },
},
{
config: { auth: Auth.header("x-custom-key", "custom") },
env: { DASHSCOPE_API_KEY: "primary" },
headers: { "x-custom-key": "custom" },
},
]) {
const provider = Alibaba.configure({ region: "ap-southeast-1", ...item.config })
for (const api of ["chat", "messages", "responses"] as const) {
const request = LLM.request({ model: provider[api]("qwen3.8-max") })
const headers = yield* request.model.route.auth
.apply({ request, method: "POST", url: "https://fixture", body: "{}", headers: Headers.empty })
.pipe(Effect.provide(ConfigProvider.layer(ConfigProvider.fromEnv({ env: item.env }))))
expect(headers).toEqual(expect.objectContaining(item.headers))
}
}
}),
)
it.effect("Alibaba keeps native reasoning controls and future efforts on their selected API", () =>
Effect.gen(function* () {
const provider = Alibaba.configure({ region: "ap-southeast-1", apiKey: "fixture" })
for (const effort of ["none", "minimal", "low", "medium", "high", "xhigh", "max", "future-effort"]) {
const chat = yield* compileRequest(
LLM.request({ model: provider.chat("qwen3.8-max"), providerOptions: { reasoningEffort: effort } }),
)
const messages = yield* compileRequest(
LLM.request({ model: provider.messages("qwen3.8-max"), providerOptions: { effort } }),
)
const responses = yield* compileRequest(
LLM.request({ model: provider.responses("qwen3.8-max"), providerOptions: { reasoningEffort: effort } }),
)
expect(chat.body.reasoning_effort).toBe(effort)
expect(messages.body.output_config).toEqual({ effort })
expect(responses.body.reasoning).toEqual({ effort })
for (const result of [chat, messages, responses]) {
expect(result.body.thinking).toBeUndefined()
expect(result.body.enable_thinking).toBeUndefined()
}
}
for (const enableThinking of [true, false]) {
const chat = yield* compileRequest(
LLM.request({
model: provider.chat("qwen3.7-plus"),
tools: [tool],
generation: { maxTokens: 1234, topK: 20 },
providerOptions: {
enableThinking,
thinkingBudget: 512,
preserveThinking: false,
clearThinking: false,
toolStream: false,
parallelToolCalls: false,
repetitionPenalty: 1.1,
responseFormat: { type: "json_object" },
enableSearch: true,
searchOptions: { forced_search: true, search_strategy: "future-strategy", enable_search_extension: false },
},
}),
)
expect(chat.body).toMatchObject({
enable_thinking: enableThinking,
thinking_budget: 512,
preserve_thinking: false,
clear_thinking: false,
tool_stream: false,
parallel_tool_calls: false,
repetition_penalty: 1.1,
max_completion_tokens: 1234,
top_k: 20,
response_format: { type: "json_object" },
enable_search: true,
search_options: { forced_search: true, search_strategy: "future-strategy", enable_search_extension: false },
})
expect(chat.body.max_tokens).toBeUndefined()
expect(chat.body.tools).toEqual([
expect.objectContaining({ function: expect.not.objectContaining({ strict: expect.anything() }) }),
])
const responses = yield* compileRequest(
LLM.request({
model: provider.responses("qwen3.7-plus"),
providerOptions: { enableThinking, store: false, previousResponseId: "resp_previous" },
}),
)
expect(responses.body).toMatchObject({
enable_thinking: enableThinking,
store: false,
previous_response_id: "resp_previous",
})
}
for (const thinking of [
{ type: "enabled" },
{ type: "disabled" },
{ type: "enabled", budgetTokens: 512 },
{ type: "future", budget_tokens: 4096 },
]) {
const messages = yield* compileRequest(
LLM.request({ model: provider.messages("qwen3.7-plus"), providerOptions: { thinking } }),
)
expect(messages.body.thinking).toEqual({
type: thinking.type,
budget_tokens: thinking.budgetTokens ?? thinking.budget_tokens,
})
}
}),
)
it.effect("Alibaba validates malformed options before execution", () =>
Effect.gen(function* () {
const provider = Alibaba.configure({ region: "ap-southeast-1", apiKey: "fixture" })
for (const [api, providerOptions] of [
["chat", { preserveThinking: "false" }],
["messages", { thinking: { type: "enabled", budgetTokens: "512" } }],
["responses", { enableThinking: "false" }],
] as const) {
const model = provider[api]("qwen3.8-max").route.with({ providerOptions }).model({ id: "qwen3.8-max" })
const error = yield* compileRequest(LLM.request({ model })).pipe(Effect.flip)
expect(error.reason._tag).toBe("InvalidRequest")
}
}),
)
it.effect("Alibaba preserves unsigned and signed Messages thinking without a budget requirement", () =>
Effect.gen(function* () {
const result = yield* compileRequest(
LLM.request({
model: Alibaba.configure({ region: "ap-southeast-1", apiKey: "fixture" }).messages("qwen3.8-max"),
providerOptions: { thinking: { type: "enabled" }, effort: "low" },
messages: [
Message.user("Hello"),
Message.assistant([
ReasoningPart.make({ type: "reasoning", text: "unsigned" }),
ReasoningPart.make({
type: "reasoning",
text: "signed",
providerMetadata: { alibaba: { signature: "opaque" } },
}),
]),
],
}),
)
expect(result.body.thinking).toEqual({ type: "enabled" })
expect(result.body.messages).toEqual(
expect.arrayContaining([
expect.objectContaining({
role: "assistant",
content: [
{ type: "thinking", thinking: "unsigned", signature: "" },
{ type: "thinking", thinking: "signed", signature: "opaque" },
],
}),
]),
)
}),
)
it.effect("Alibaba serializes Messages configuration, per-request controls, and final HTTP overlays", () =>
LLMClient.generate(
LLM.request({
model: Alibaba.configure({
baseURL: "https://gateway.example/v1",
apiKey: "fixture",
providerOptions: { thinking: { type: "enabled", budgetTokens: 512 }, effort: "high" },
}).messages("qwen3.8-max"),
prompt: "Hello",
providerOptions: { effort: "low" },
http: { body: { output_config: { effort: "medium" }, extension: true } },
}),
).pipe(
Effect.provide(
dynamicResponse((input) =>
Effect.sync(() => {
expect(input.request.url).toBe("https://gateway.example/v1/messages")
expect(input.request.headers.authorization).toBe("Bearer fixture")
expect(input.request.headers["anthropic-version"]).toBe("2023-06-01")
expect(JSON.parse(input.text)).toMatchObject({
thinking: { type: "enabled", budget_tokens: 512 },
output_config: { effort: "medium" },
extension: true,
})
return input.respond(
sseEvents(
{
type: "message_start",
message: { id: "msg_fixture", content: [], usage: { input_tokens: 1, output_tokens: 0 } },
},
{ type: "message_delta", delta: { stop_reason: "end_turn" }, usage: { output_tokens: 1 } },
{ type: "message_stop" },
),
{ headers: { "content-type": "text/event-stream" } },
)
}),
),
),
),
)
it.effect("Alibaba Responses lowers hosted tools and named selection using HTTP even with a WebSocket executor", () =>
Effect.gen(function* () {
const request = LLM.request({
model: Alibaba.configure({ region: "ap-southeast-1", apiKey: "fixture" }).responses("qwen3.8-max"),
tools: [Alibaba.webSearch(), Alibaba.webExtractor(), Alibaba.codeInterpreter(), tool],
prompt: "Hello",
})
const compiled = yield* compileRequest(request)
expect(compiled.body.tools).toMatchObject([
{ type: "web_search" },
{ type: "web_extractor" },
{ type: "code_interpreter" },
{ type: "function", name: "lookup" },
])
const named = yield* compileRequest(
LLMRequest.update(request, { tools: [tool], toolChoice: { type: "tool", name: "lookup" } }),
)
expect(named.body.tool_choice).toEqual({
type: "allowed_tools",
mode: "required",
tools: [{ type: "function", name: "lookup" }],
})
const response = yield* LLMClient.generate(request, {
webSocket: { execute: () => Effect.die("Unexpected WebSocket") },
})
expect(response.toolCalls).toMatchObject([
{
name: "web_extractor",
providerExecuted: true,
input: { urls: ["https://example.com"], goal: "Read the page" },
},
])
const replay = yield* compileRequest(
LLMRequest.update(request, { messages: [...request.messages, response.message, Message.user("Continue")] }),
)
expect(replay.body.input).toEqual(
expect.arrayContaining([
expect.objectContaining({
type: "web_extractor_call",
id: "extract_1",
urls: ["https://example.com"],
goal: "Read the page",
result: { text: "fixture" },
}),
]),
)
}).pipe(
Effect.provide(
dynamicResponse((input) =>
Effect.sync(() => {
expect(input.request.url).toBe("https://dashscope-intl.aliyuncs.com/compatible-mode/v1/responses")
return input.respond(
sseEvents(
{
type: "response.output_item.done",
output_index: 0,
item: {
type: "web_extractor_call",
id: "extract_1",
status: "completed",
urls: ["https://example.com"],
goal: "Read the page",
result: { text: "fixture" },
},
},
{ type: "response.completed", response: {} },
),
{ headers: { "content-type": "text/event-stream" } },
)
}),
),
),
),
)
@@ -1458,13 +1458,7 @@ describe("Bedrock Converse route", () => {
expect(headers.get("authorization")).toContain("Credential=AKIACHAINEXAMPLE/")
expect(headers.get("authorization")).toContain("/ap-southeast-2/bedrock/aws4_request")
}
expect(() => AmazonBedrock.configure({ auth: "sigv4", apiKey: "k" })).toThrow(
expect.objectContaining({
_tag: "ProviderConfiguration",
provider: "amazon-bedrock",
message: "Amazon Bedrock SigV4 auth does not accept apiKey",
}),
)
expect(() => AmazonBedrock.configure({ auth: "sigv4", apiKey: "k" })).toThrow("does not accept apiKey")
}).pipe(
withProcessEnv({
...noAmbientAWS,
@@ -4,7 +4,7 @@ import { HttpClientRequest } from "effect/unstable/http"
import { LLM, Message } from "../../src/index.js"
import { AmazonBedrockMantle } from "../../src/providers.js"
import { model } from "../../src/providers/amazon-bedrock/mantle.js"
import { OpenResponses } from "../../src/protocols/open-responses.js"
import { OpenAIResponses } from "../../src/protocols/openai-responses.js"
import { compileRequest, LLMClient } from "../../src/route/client.js"
import { it } from "../lib/effect.js"
import { withProcessEnv } from "../lib/env.js"
@@ -25,7 +25,7 @@ describe("Amazon Bedrock Mantle provider", () => {
expect(provider.model).toBe(provider.responses)
expect(AmazonBedrockMantle.model).toBe(AmazonBedrockMantle.responsesModel)
expect(model).toBe(AmazonBedrockMantle.responsesModel)
expect(provider.model("openai.gpt-oss-120b").route.transport).toBe(OpenResponses.httpTransport)
expect(provider.model("openai.gpt-oss-120b").route.transport).toBe(OpenAIResponses.httpTransport)
const chat = yield* compileRequest(LLM.request({ model: provider.chat("openai.gpt-oss-120b"), prompt: "Hi" }))
const responses = yield* compileRequest(
LLM.request({ model: provider.model("openai.gpt-oss-120b"), prompt: "Hi" }),
@@ -38,7 +38,7 @@ describe("Amazon Bedrock Mantle provider", () => {
})
expect(responses).toMatchObject({
route: "bedrock-mantle-responses",
protocol: "open-responses",
protocol: "openai-responses",
body: { model: "openai.gpt-oss-120b", store: false },
})
expect(provider.model("openai.gpt-oss-120b").route.providerMetadataKey).toBe("mantle")
@@ -178,7 +178,7 @@ describe("Amazon Bedrock Mantle provider", () => {
const recorded = recordedTests({
prefix: "bedrock-mantle",
provider: "amazon-bedrock",
protocol: "open-responses",
protocol: "openai-responses",
requires: ["AWS_BEARER_TOKEN_BEDROCK"],
metadata: { model: "openai.gpt-oss-120b" },
})
@@ -23,7 +23,7 @@ it.effect("conversation lowering excludes generation settings and tool definitio
instructions: "Keep the context",
input: [
{ role: "user", content: [{ type: "input_text", text: "hello" }] },
{ type: "message", role: "assistant", status: "completed", content: [{ type: "output_text", text: "hi" }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "hi" }] },
],
})
}),
@@ -378,11 +378,7 @@ describe("Google Vertex providers", () => {
test("rejects tuned Gemini models in express mode", () => {
expect(() => GoogleVertex.configure({ apiKey: "fixture" }).model("endpoints/1234567890")).toThrow(
expect.objectContaining({
_tag: "ProviderConfiguration",
provider: "google-vertex",
message: "Google Vertex tuned models do not support Express Mode API keys",
}),
"Google Vertex tuned models do not support Express Mode API keys",
)
})
})
@@ -1,78 +0,0 @@
import { expect } from "bun:test"
import { Effect, Schema } from "effect"
import { LLM, LLMClient } from "../../src/index.js"
import { OpenResponses } from "../../src/protocols/open-responses.js"
import { Meta } from "../../src/providers/index.js"
import { configure } from "../../src/providers/openai-compatible-responses.js"
import { it } from "../lib/effect.js"
import { fixedResponse } from "../lib/http.js"
import { sseEvents } from "../lib/sse.js"
const decodeEvent = Schema.decodeUnknownEffect(OpenResponses.protocol.stream.event)
it.effect("normalizes flat errors in shared SSE and WebSocket decoding", () =>
Effect.gen(function* () {
const frame = {
type: "error",
sequence_number: 4,
code: "server_shutting_down",
message: "Server is shutting down. Please retry your request.",
param: null,
}
for (const decode of [decodeEvent, OpenResponses.decodeChannelEvent]) {
const event = yield* decode(JSON.stringify(frame))
expect(event).toEqual({
type: "error",
sequence_number: 4,
error: { code: frame.code, message: frame.message, param: null },
})
for (const unchanged of [
event,
{ type: "error" },
{
type: "response.failed",
response: { id: "resp_failed", error: { code: "server_error", message: "Internal server error" } },
},
{ type: "response.output_text.delta", item_id: "msg_text", delta: "Hello" },
]) {
expect(yield* decode(JSON.stringify(unchanged))).toEqual(unchanged)
}
}
}),
)
it.effect("continues to normalize untyped xAI WebSocket errors", () =>
Effect.gen(function* () {
const frame = { error: { type: "api_error", message: "gRPC error: Response with id=resp_missing not found" } }
expect(yield* OpenResponses.decodeChannelEvent(JSON.stringify(frame))).toEqual({ ...frame, type: "error" })
}),
)
it.effect("retains classification and original error bodies through Meta and generic Responses routes", () =>
Effect.gen(function* () {
const raw = `{
"type": "error",
"sequence_number": 4,
"code": "server_shutting_down",
"message": "Server is shutting down. Please retry your request.",
"param": null,
"diagnostic": "retain-original-frame"
}`
for (const model of [
Meta.configure({ apiKey: "fixture" }).responses("muse-spark-1.3"),
configure({ apiKey: "fixture", provider: "gateway", baseURL: "https://responses.example.test/v1" }).model(
"example-model",
),
]) {
const error = yield* LLMClient.generate(LLM.request({ model, prompt: "Hello" })).pipe(
Effect.provide(fixedResponse(sseEvents(raw.replaceAll("\n", "\ndata: ")))),
Effect.flip,
)
expect(error.reason._tag).toBe("ProviderInternal")
expect(error.message).toBe("server_shutting_down: Server is shutting down. Please retry your request.")
expect(error.reason.body).toBe(raw)
expect(error.reason.http?.status).toBe(200)
}
}),
)
@@ -1,120 +0,0 @@
import { describe, expect } from "bun:test"
import { Effect } from "effect"
import { LLM, LLMEvent, Message } from "../../src/index.js"
import { OpenAI } from "../../src/providers.js"
import { configure } from "../../src/providers/openai-compatible-responses.js"
import { compileRequest, LLMClient } from "../../src/route/client.js"
import { it } from "../lib/effect.js"
import { fixedResponse } from "../lib/http.js"
import { sseEvents } from "../lib/sse.js"
for (const model of [
OpenAI.configure({ apiKey: "test-key" }).responses("example-model"),
configure({ apiKey: "test-key", baseURL: "https://responses.example.test/v1" }).model("example-model"),
]) {
describe(`${model.route.protocol} message replay`, () => {
const key = model.route.providerMetadataKey ?? "openresponses"
it.effect("marks assistant text completed regardless of stored status", () =>
Effect.gen(function* () {
const prepared = yield* compileRequest(
LLM.request({
model,
messages: [
...[undefined, "in_progress", "incomplete", "completed"].map((status, index) =>
Message.make({
role: "assistant",
providerMetadata: { [key]: { status } },
content: [
{
type: "text",
text: `Saved ${index}`,
providerMetadata: { [key]: { itemId: `msg_${index}`, phase: "commentary", status } },
},
{
type: "text",
text: `Final ${index}`,
providerMetadata: { [key]: { itemId: `msg_final_${index}`, phase: "final_answer", status } },
},
],
}),
),
Message.make({
role: "user",
content: [{ type: "text", text: "Continue" }],
providerMetadata: { [key]: { status: "incomplete" } },
}),
],
}),
)
expect(prepared.body.input).toEqual([
...[0, 1, 2, 3].flatMap((index) => [
{
type: "message",
role: "assistant",
id: `msg_${index}`,
phase: "commentary",
status: "completed",
content: [{ type: "output_text", text: `Saved ${index}` }],
},
{
type: "message",
role: "assistant",
id: `msg_final_${index}`,
phase: "final_answer",
status: "completed",
content: [{ type: "output_text", text: `Final ${index}` }],
},
]),
{ role: "user", status: "incomplete", content: [{ type: "input_text", text: "Continue" }] },
])
}),
)
it.effect("replays truncated text as completed while retaining the response finish reason", () =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(LLM.request({ model, prompt: "Respond" })).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{
type: "response.output_item.added",
item: { type: "message", id: "msg_partial", status: "in_progress" },
},
{ type: "response.output_text.delta", item_id: "msg_partial", delta: "The next step is" },
{
type: "response.output_item.done",
item: {
type: "message",
id: "msg_partial",
status: "incomplete",
content: [{ type: "output_text", text: "The next step is" }],
},
},
{
type: "response.incomplete",
response: { status: "incomplete", incomplete_details: { reason: "max_output_tokens" } },
},
),
),
),
)
expect(response.finishReason.normalized).toBe("length")
expect(response.events.filter(LLMEvent.is.textEnd)).toHaveLength(1)
const prepared = yield* compileRequest(
LLM.request({ model, messages: [response.message, Message.user("Continue")] }),
)
expect(prepared.body.input).toEqual([
{
type: "message",
role: "assistant",
id: "msg_partial",
status: "completed",
content: [{ type: "output_text", text: "The next step is" }],
},
{ role: "user", content: [{ type: "input_text", text: "Continue" }] },
])
}),
)
})
}
@@ -91,7 +91,7 @@ describe("Open Responses-compatible route", () => {
expect(prepared.body.input).toEqual([
{ role: "user", content: [{ type: "input_text", text: "Before." }] },
{ role: "developer", content: "Operator update." },
{ type: "message", role: "assistant", status: "completed", content: [{ type: "output_text", text: "After." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "After." }] },
])
}),
)
@@ -299,27 +299,23 @@ describe("Open Responses-compatible route", () => {
type: "message",
id: "history_1",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Kept." }],
},
{
type: "message",
id: `history_${"a".repeat(64)}`,
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Long." }],
},
{
type: "message",
id: "provider_value/with+symbols",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Opaque." }],
},
{
type: "message",
role: "assistant",
status: "completed",
content: [
{ type: "output_text", text: "No suffix." },
{ type: "output_text", text: "No prefix." },
@@ -860,7 +856,6 @@ describe("Open Responses-compatible route", () => {
type: "message",
id: "msg_refusal",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "I can't help with that." }],
},
])
@@ -171,7 +171,7 @@ describe("OpenAI Responses WebSocket recorded", () => {
instructions: "Follow the user's exact reply instruction.",
input: [
{ role: "user", content: [{ type: "input_text", text: "Reply exactly: Alpha." }] },
{ role: "assistant", status: "completed", content: [{ type: "output_text", text: "Alpha." }] },
{ role: "assistant", content: [{ type: "output_text", text: "Alpha." }] },
{ role: "user", content: [{ type: "input_text", text: "Reply exactly: Beta." }] },
],
})
@@ -208,7 +208,7 @@ describe("OpenAI Responses WebSocket recorded", () => {
instructions: "Follow the user's exact reply instruction.",
input: [
{ role: "user", content: [{ type: "input_text", text: "Reply exactly: Ready." }] },
{ role: "assistant", status: "completed", content: [{ type: "output_text", text: "Ready." }] },
{ role: "assistant", content: [{ type: "output_text", text: "Ready." }] },
{ role: "user", content: [{ type: "input_text", text: "Reply exactly: Recovered." }] },
],
})
@@ -1,5 +1,5 @@
import { describe, expect } from "bun:test"
import { ConfigProvider, Effect, Layer, Ref, Schema, Stream } from "effect"
import { ConfigProvider, Effect, Layer, Ref, Stream } from "effect"
import { Headers, HttpClientRequest } from "effect/unstable/http"
import {
LLM,
@@ -30,7 +30,6 @@ import * as Azure from "../../src/providers/azure.js"
import * as OpenAI from "../../src/providers/openai.js"
import * as XAI from "../../src/providers/xai.js"
import * as OpenAIResponses from "../../src/protocols/openai-responses.js"
import { OpenResponses } from "../../src/protocols/open-responses.js"
import { OpenResponsesContinuation } from "../../src/protocols/open-responses-continuation.js"
import * as ProviderShared from "../../src/protocols/shared.js"
import { continuationRequest, nativeOpenAIResponsesContinuation } from "../continuation-scenarios.js"
@@ -70,39 +69,14 @@ const baseChannelDriver = (message: string): WebSocketChannelDriver => ({
},
})
/** Classifies error frames the way the production channel does, so recovery can read the canonical reason. */
const classifyingChannelDriver = (message: string): WebSocketChannelDriver => {
const base = baseChannelDriver(message)
const decodeEvent = Schema.decodeUnknownSync(OpenResponses.protocol.stream.event)
return {
...base,
observe: (create, frame) =>
base.observe(create, frame).pipe(
Effect.map((observation) =>
observation.type === "provider-failure"
? {
...observation,
error: OpenResponses.providerFailure(decodeEvent(frame), "stream error", frame),
}
: observation,
),
),
}
}
const continuationDriver = (
request: Readonly<Record<string, unknown>>,
base = baseChannelDriver,
continuation?: OpenResponsesContinuation.Shape,
) => {
const continuationDriver = (request: Readonly<Record<string, unknown>>) => {
const message = ProviderShared.encodeJson(request)
return OpenResponsesContinuation.driver({
id: "openai-responses",
name: "OpenAI Responses",
request,
message,
base: base(message),
continuation,
base: baseChannelDriver(message),
})
}
@@ -411,7 +385,7 @@ describe("OpenAI Responses route", () => {
expect(prepared.body.input).toEqual([
{ role: "user", content: [{ type: "input_text", text: "Before." }] },
{ role: "developer", content: "Operator update." },
{ type: "message", role: "assistant", status: "completed", content: [{ type: "output_text", text: "After." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "After." }] },
])
}),
)
@@ -586,54 +560,52 @@ describe("OpenAI Responses route", () => {
)
it.effect("continues a streamed tool call with only the new tool output", () =>
Effect.forEach([undefined, []], (output) =>
Effect.gen(function* () {
const firstRequest = {
type: "response.create",
model: "gpt-5.2",
store: false,
input: [{ role: "user", content: [{ type: "input_text", text: "Weather?" }] }],
}
const first = continuationDriver(firstRequest)
const firstCreate = yield* first.create(undefined)
Effect.gen(function* () {
const firstRequest = {
type: "response.create",
model: "gpt-5.2",
store: false,
input: [{ role: "user", content: [{ type: "input_text", text: "Weather?" }] }],
}
const first = continuationDriver(firstRequest)
const firstCreate = yield* first.create(undefined)
yield* first.observe(
firstCreate,
ProviderShared.encodeJson({
type: "response.output_item.done",
item: {
type: "function_call",
id: "fc_1",
status: "completed",
call_id: "call_1",
name: "weather",
arguments: '{ "city": "Paris" }',
},
}),
)
const saved = checkpoint(
yield* first.observe(
firstCreate,
ProviderShared.encodeJson({
type: "response.output_item.done",
item: {
type: "function_call",
id: "fc_1",
status: "completed",
call_id: "call_1",
name: "weather",
arguments: '{ "city": "Paris" }',
},
}),
)
const saved = checkpoint(
yield* first.observe(
firstCreate,
ProviderShared.encodeJson({ type: "response.completed", response: { id: "resp_1", output } }),
),
)
const second = continuationDriver({
...firstRequest,
input: [
...firstRequest.input,
{ type: "function_call", call_id: "call_1", name: "weather", arguments: '{"city":"Paris"}' },
{ type: "function_call_output", call_id: "call_1", output: '{"temperature":22}' },
],
})
ProviderShared.encodeJson({ type: "response.completed", response: { id: "resp_1" } }),
),
)
const second = continuationDriver({
...firstRequest,
input: [
...firstRequest.input,
{ type: "function_call", call_id: "call_1", name: "weather", arguments: '{"city":"Paris"}' },
{ type: "function_call_output", call_id: "call_1", output: '{"temperature":22}' },
],
})
const create = yield* second.create(saved)
const create = yield* second.create(saved)
expect(create.mode).toBe("incremental")
expect(ProviderShared.decodeJson(create.message)).toMatchObject({
previous_response_id: "resp_1",
input: [{ type: "function_call_output", call_id: "call_1", output: '{"temperature":22}' }],
})
}),
),
expect(create.mode).toBe("incremental")
expect(ProviderShared.decodeJson(create.message)).toMatchObject({
previous_response_id: "resp_1",
input: [{ type: "function_call_output", call_id: "call_1", output: '{"temperature":22}' }],
})
}),
)
it.effect("continues a tool call from authoritative completed response output", () =>
@@ -687,47 +659,45 @@ describe("OpenAI Responses route", () => {
)
it.effect("continues a promoted steer after assistant output with response-only text metadata", () =>
Effect.forEach([undefined, []], (output) =>
Effect.gen(function* () {
const firstInput = [{ role: "user", content: [{ type: "input_text", text: "First" }] }]
const first = continuationDriver({ type: "response.create", model: "gpt-5.2", store: false, input: firstInput })
const create = yield* first.create(undefined)
Effect.gen(function* () {
const firstInput = [{ role: "user", content: [{ type: "input_text", text: "First" }] }]
const first = continuationDriver({ type: "response.create", model: "gpt-5.2", store: false, input: firstInput })
const create = yield* first.create(undefined)
yield* first.observe(
create,
ProviderShared.encodeJson({
type: "response.output_item.done",
item: {
type: "message",
id: "msg_1",
status: "completed",
role: "assistant",
content: [{ type: "output_text", text: "Hello", annotations: [], logprobs: [] }],
},
}),
)
const saved = checkpoint(
yield* first.observe(
create,
ProviderShared.encodeJson({
type: "response.output_item.done",
item: {
type: "message",
id: "msg_1",
status: "completed",
role: "assistant",
content: [{ type: "output_text", text: "Hello", annotations: [], logprobs: [] }],
},
}),
)
const saved = checkpoint(
yield* first.observe(
create,
ProviderShared.encodeJson({ type: "response.completed", response: { id: "resp_1", output } }),
),
)
const steer = { role: "user", content: [{ type: "input_text", text: "Actually, be brief" }] }
const next = continuationDriver({
type: "response.create",
model: "gpt-5.2",
store: false,
input: [...firstInput, { role: "assistant", content: [{ type: "output_text", text: "Hello" }] }, steer],
})
ProviderShared.encodeJson({ type: "response.completed", response: { id: "resp_1" } }),
),
)
const steer = { role: "user", content: [{ type: "input_text", text: "Actually, be brief" }] }
const next = continuationDriver({
type: "response.create",
model: "gpt-5.2",
store: false,
input: [...firstInput, { role: "assistant", content: [{ type: "output_text", text: "Hello" }] }, steer],
})
const continued = yield* next.create(saved)
const continued = yield* next.create(saved)
expect(continued.mode).toBe("incremental")
expect(ProviderShared.decodeJson(continued.message)).toMatchObject({
previous_response_id: "resp_1",
input: [steer],
})
}),
),
expect(continued.mode).toBe("incremental")
expect(ProviderShared.decodeJson(continued.message)).toMatchObject({
previous_response_id: "resp_1",
input: [steer],
})
}),
)
it.effect("continues streamed reasoning when completion re-encrypts the same item", () =>
@@ -882,105 +852,6 @@ describe("OpenAI Responses route", () => {
}),
)
it.effect("retries an incremental send in full when the provider rejects it without a code", () =>
Effect.gen(function* () {
const firstRequest = {
type: "response.create",
model: "gpt-5.2",
store: false,
input: [{ role: "user", content: [{ type: "input_text", text: "First" }] }],
}
const first = continuationDriver(firstRequest, classifyingChannelDriver)
const saved = checkpoint(
yield* first.observe(
yield* first.create(undefined),
ProviderShared.encodeJson({ type: "response.completed", response: { id: "resp_1" } }),
),
)
const second = continuationDriver(
{
...firstRequest,
input: [...firstRequest.input, { role: "user", content: [{ type: "input_text", text: "Second" }] }],
},
classifyingChannelDriver,
)
// Codex reports a stale previous_response_id as a plain invalid_request_error.
const stale = ProviderShared.encodeJson({
type: "error",
error: { type: "invalid_request_error", message: "Invalid `previous_response_id`." },
})
const incremental = yield* second.create(saved)
expect(incremental.mode).toBe("incremental")
expect(yield* second.observe(incremental, stale)).toMatchObject({ type: "rejected", recovery: "retry-full" })
// A full send has no continuation to blame, so the same error stays a provider failure.
const full = yield* second.create(undefined)
expect(yield* second.observe(full, stale)).toMatchObject({ type: "provider-failure" })
// A classified failure keeps its runner-owned recovery instead of resending the whole context.
const overflow = ProviderShared.encodeJson({
type: "error",
error: { type: "invalid_request_error", code: "context_length_exceeded", message: "Too long" },
})
expect(yield* second.observe(yield* second.create(saved), overflow)).toMatchObject({
type: "provider-failure",
error: { reason: { _tag: "InvalidRequest", classification: "context-overflow" } },
})
// A retryable failure stays one: the runner retries it, and the transport has already dropped the
// checkpoint, so that retry is a full send. xAI reports every rejection this way.
const internal = ProviderShared.encodeJson({
type: "error",
error: { type: "api_error", message: "gRPC error: Response with id=resp_1 not found" },
})
expect(yield* second.observe(yield* second.create(saved), internal)).toMatchObject({
type: "provider-failure",
error: { reason: { _tag: "ProviderInternal" } },
})
}),
)
it.effect("shapes the incremental send with the route continuation", () =>
Effect.gen(function* () {
const firstRequest = {
type: "response.create",
model: "grok-4.6",
store: true,
instructions: "You are terse.",
input: [{ role: "user", content: [{ type: "input_text", text: "First" }] }],
}
const secondRequest = {
...firstRequest,
input: [...firstRequest.input, { role: "user", content: [{ type: "input_text", text: "Second" }] }],
}
const saved = checkpoint(
yield* continuationDriver(firstRequest).observe(
yield* continuationDriver(firstRequest).create(undefined),
ProviderShared.encodeJson({ type: "response.completed", response: { id: "resp_1" } }),
),
)
const trimmed = yield* continuationDriver(
secondRequest,
baseChannelDriver,
({ instructions: _, ...rest }) => rest,
).create(saved)
expect(trimmed.mode).toBe("incremental")
expect(JSON.parse(trimmed.message)).toEqual({
type: "response.create",
model: "grok-4.6",
store: true,
previous_response_id: "resp_1",
input: [{ role: "user", content: [{ type: "input_text", text: "Second" }] }],
})
// Declining the continuation sends the step in full and never sends a previous_response_id.
const declined = yield* continuationDriver(secondRequest, baseChannelDriver, () => undefined).create(saved)
expect(declined.mode).toBe("full")
expect(JSON.parse(declined.message)).toEqual(secondRequest)
}),
)
it.effect("builds WebSocket and HTTP fallback from the same final request", () =>
Effect.gen(function* () {
const attempts = yield* Ref.make(0)
@@ -2179,7 +2050,6 @@ describe("OpenAI Responses route", () => {
type: "message",
id: "msg_refusal",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "I can't help with that." }],
phase: "final_answer",
},
@@ -2262,7 +2132,6 @@ describe("OpenAI Responses route", () => {
type: "message",
id: "msg_commentary",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Checking." }],
phase: "commentary",
},
@@ -2270,7 +2139,6 @@ describe("OpenAI Responses route", () => {
type: "message",
id: "msg_final",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Finished." }],
phase: "final_answer",
},
@@ -2278,7 +2146,6 @@ describe("OpenAI Responses route", () => {
type: "message",
id: "msg_null",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Unclassified." }],
phase: null,
},
@@ -3409,19 +3276,14 @@ describe("OpenAI Responses route", () => {
)
expect(prepared.body.input).toEqual([
{
type: "message",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Before." }],
},
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "Before." }] },
{
type: "reasoning",
id: "rs_1",
encrypted_content: "encrypted-state",
summary: [{ type: "summary_text", text: "Checked order." }],
},
{ type: "message", role: "assistant", status: "completed", content: [{ type: "output_text", text: "After." }] },
{ type: "message", role: "assistant", content: [{ type: "output_text", text: "After." }] },
])
}),
)
@@ -3685,14 +3547,12 @@ describe("OpenAI Responses route", () => {
type: "message",
id: "history_1",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Hello" }],
},
{
type: "message",
id: `message_${"a".repeat(64)}`,
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "World" }],
},
{
@@ -1,361 +0,0 @@
import { describe, expect } from "bun:test"
import { Effect, Schema } from "effect"
import { LLM, LLMEvent, Message } from "../../src/index.js"
import { configure } from "../../src/providers/organization-routes.js"
import { provider } from "../../src/providers/openai-compatible-responses.js"
import { LLMClient } from "../../src/route.js"
import { compileRequest } from "../../src/route/client.js"
import { it } from "../lib/effect.js"
import { dynamicResponse, fixedResponse } from "../lib/http.js"
import { sseEvents } from "../lib/sse.js"
const model = configure({
apiKey: "session-test",
baseURL: "https://console.example.test/inference/route/openai/v1",
provider: "opencode-routes-org_test",
headers: { "x-opencode-org-id": "org_test" },
}).model("route/coding")
const fixtures = [
{
protocol: "openai-responses",
content: [
{
type: "reasoning",
id: "rs_native",
summary: [{ type: "summary_text", text: "Need a lookup" }],
encrypted_content: "opaque-openai-reasoning",
},
{ type: "message", id: "msg_native", role: "assistant", content: [{ type: "output_text", text: "Checking." }] },
{
type: "function_call",
id: "fc_native_lookup",
call_id: "call_lookup",
name: "lookup",
arguments: '{"query":"weather"}',
},
{ type: "function_call", id: "fc_native_clock", call_id: "call_clock", name: "clock", arguments: "{}" },
],
},
{
protocol: "google",
content: [
{ text: "Need a lookup", thought: true },
{ text: "Checking." },
{ functionCall: { name: "lookup", args: { query: "weather" } }, thoughtSignature: "opaque-google-signature" },
{ functionCall: { name: "clock", args: {} } },
],
},
{
protocol: "anthropic-messages",
content: [
{ type: "thinking", thinking: "Need a lookup", signature: "opaque-anthropic-signature" },
{ type: "text", text: "Checking." },
{ type: "tool_use", id: "call_lookup", name: "lookup", input: { query: "weather" } },
{ type: "tool_use", id: "call_clock", name: "clock", input: {} },
],
},
{
protocol: "openai-chat",
content: [
{
role: "assistant",
reasoning_content: "Need a lookup",
content: "Checking.",
tool_calls: [
{ id: "call_lookup", type: "function", function: { name: "lookup", arguments: '{"query":"weather"}' } },
{ id: "call_clock", type: "function", function: { name: "clock", arguments: "{}" } },
],
},
],
},
]
const finalText = sseEvents(
{
type: "response.output_item.done",
item: { type: "message", id: "msg_final", content: [{ type: "output_text", text: "Sunny." }] },
},
{ type: "response.completed", response: { id: "resp_final" } },
)
describe("organization routes", () => {
it.effect("uses the Responses route endpoint with stateless portable options", () =>
Effect.gen(function* () {
const response = yield* LLMClient.generate(
LLM.request({
model,
system: "Help with code.",
prompt: "Hello",
promptCacheKey: "session-cache",
providerOptions: {
store: true,
include: ["reasoning.encrypted_content"],
previousResponseId: "resp_old",
reasoningEffort: "low",
reasoningSummary: "auto",
serviceTier: "priority",
},
}),
).pipe(
Effect.provide(
dynamicResponse((input) =>
Effect.gen(function* () {
expect(input.request.url).toBe("https://console.example.test/inference/route/openai/v1/responses")
expect(input.request.headers.authorization).toBe("Bearer session-test")
expect(input.request.headers["x-opencode-org-id"]).toBe("org_test")
expect(yield* Schema.decodeUnknownEffect(Schema.fromJsonString(Schema.Unknown))(input.text)).toEqual({
model: "route/coding",
input: [{ role: "user", content: [{ type: "input_text", text: "Hello" }] }],
instructions: "Help with code.",
stream: true,
store: false,
provider_options: { "openai-responses": { include: ["reasoning.encrypted_content"] } },
reasoning: { effort: "low" },
})
return input.respond(finalText, { headers: { "content-type": "text/event-stream" } })
}),
),
),
)
expect(response.text).toBe("Sunny.")
}),
)
it.effect("omits an unknown output limit", () =>
Effect.gen(function* () {
const prepared = yield* compileRequest(LLM.request({ model, prompt: "Hello", generation: { maxTokens: 0 } }))
expect(prepared.body.max_output_tokens).toBeUndefined()
}),
)
for (const fixture of fixtures) {
it.effect(`preserves ${fixture.protocol} continuation metadata through a streamed parallel tool loop`, () =>
Effect.gen(function* () {
const metadata = {
protocol: fixture.protocol,
model: "native-model",
connection_id: "conn_native",
endpoint_id: "endpoint_native",
content: fixture.content,
group_id: "resp_tools",
}
const marker = { group_id: "resp_tools" }
const items = [
{
type: "reasoning",
id: "rs_1",
summary: [{ type: "summary_text", text: "Need a lookup" }],
provider_metadata: metadata,
},
{
type: "message",
id: "msg_1",
role: "assistant",
content: [{ type: "output_text", text: "Checking." }],
provider_metadata: marker,
},
{
type: "function_call",
id: "fc_lookup",
call_id: "call_lookup",
name: "lookup",
arguments: '{"query":"weather"}',
provider_metadata: marker,
},
{
type: "function_call",
id: "fc_clock",
call_id: "call_clock",
name: "clock",
arguments: "{}",
provider_metadata: marker,
},
]
const first = yield* LLMClient.generate(LLM.request({ model, prompt: "Check weather and time." })).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{
type: "response.output_item.added",
output_index: 0,
item: { type: "reasoning", id: "rs_1", summary: [] },
},
{
type: "response.reasoning_summary_text.delta",
item_id: "rs_1",
summary_index: 0,
delta: "Need a lookup",
},
{
type: "response.output_item.added",
output_index: 1,
item: { type: "message", id: "msg_1", role: "assistant", content: [] },
},
{ type: "response.output_text.delta", item_id: "msg_1", delta: "Checking." },
{
type: "response.output_item.added",
output_index: 2,
item: {
type: "function_call",
id: "fc_lookup",
call_id: "call_lookup",
name: "lookup",
arguments: "",
},
},
{ type: "response.function_call_arguments.delta", item_id: "fc_lookup", delta: '{"query":"weather"}' },
{
type: "response.output_item.added",
output_index: 3,
item: { type: "function_call", id: "fc_clock", call_id: "call_clock", name: "clock", arguments: "" },
},
{ type: "response.function_call_arguments.delta", item_id: "fc_clock", delta: "{}" },
...items.map((item, output_index) => ({ type: "response.output_item.done", output_index, item })),
{ type: "response.completed", response: { id: "resp_tools", output: items } },
),
),
),
)
expect(first.toolCalls).toHaveLength(2)
expect(first.events.filter(LLMEvent.is.toolCall)).toHaveLength(2)
expect(
first.message.content.map((part) => part.providerMetadata?.["opencode-routes-org_test"]?.providerMetadata),
).toEqual([metadata, marker, marker, marker])
const second = yield* LLMClient.generate(
LLM.request({
model,
providerOptions: { previousResponseId: "resp_tools" },
messages: [
Message.user("Check weather and time."),
first.message,
Message.tool({ id: "call_lookup", name: "lookup", result: { weather: "sunny" } }),
Message.tool({ id: "call_clock", name: "clock", result: { time: "noon" } }),
],
}),
).pipe(
Effect.provide(
dynamicResponse((input) =>
Effect.gen(function* () {
const body = yield* Schema.decodeUnknownEffect(
Schema.fromJsonString(
Schema.Struct({
input: Schema.Array(Schema.Record(Schema.String, Schema.Unknown)),
store: Schema.Boolean,
include: Schema.optional(Schema.Array(Schema.String)),
previous_response_id: Schema.optional(Schema.String),
}),
),
)(input.text)
expect(body.store).toBe(false)
expect(body.include).toBeUndefined()
expect(body.previous_response_id).toBeUndefined()
expect(body.input.map((item) => item.type ?? item.role)).toEqual([
"user",
"reasoning",
"message",
"function_call",
"function_call",
"function_call_output",
"function_call_output",
])
expect(body.input.slice(1, 5).map((item) => item.provider_metadata)).toEqual([
metadata,
marker,
marker,
marker,
])
expect(body.input.slice(3, 5).map((item) => item.call_id)).toEqual(["call_lookup", "call_clock"])
return input.respond(finalText, { headers: { "content-type": "text/event-stream" } })
}),
),
),
)
expect(second.text).toBe("Sunny.")
}),
)
}
it.effect("preserves opaque state on an empty reasoning item", () =>
Effect.gen(function* () {
const metadata = {
protocol: "anthropic-messages",
content: [{ type: "redacted_thinking", data: "opaque" }],
group_id: "resp_empty",
}
const response = yield* LLMClient.generate(LLM.request({ model, prompt: "Think." })).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{
type: "response.output_item.done",
item: { type: "reasoning", id: "rs_empty", summary: [], provider_metadata: metadata },
},
{ type: "response.completed", response: { id: "resp_empty" } },
),
),
),
)
const replay = yield* compileRequest(LLM.request({ model, messages: [response.message] }))
expect(replay.body.input).toEqual([
{ type: "reasoning", id: "rs_empty", summary: [], encrypted_content: null, provider_metadata: metadata },
])
}),
)
it.effect("keeps opaque route metadata out of other Responses providers", () =>
Effect.gen(function* () {
const other = provider
.configure({
apiKey: "test-key",
baseURL: "https://other.example.test/v1",
provider: "opencode-routes-org_test",
})
.model("other")
const response = yield* LLMClient.generate(LLM.request({ model: other, prompt: "Hello" })).pipe(
Effect.provide(
fixedResponse(
sseEvents(
{
type: "response.output_item.done",
item: {
type: "message",
id: "msg_1",
content: [{ type: "output_text", text: "Hi" }],
provider_metadata: { group_id: "resp_1", secret: "opaque" },
},
},
{ type: "response.completed", response: { id: "resp_1" } },
),
),
),
)
expect(response.message.content[0]?.providerMetadata).toEqual({ "opencode-routes-org_test": { itemId: "msg_1" } })
const replay = yield* compileRequest(
LLM.request({
model: other,
messages: [
Message.assistant({
type: "text",
text: "Hi",
providerMetadata: {
"opencode-routes-org_test": { itemId: "msg_1", providerMetadata: { secret: "opaque" } },
},
}),
],
}),
)
expect(replay.body.input).toEqual([
{
type: "message",
id: "msg_1",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Hi" }],
},
])
expect(replay.body.include).toEqual(["reasoning.encrypted_content"])
}),
)
})
+2 -110
View File
@@ -1,18 +1,11 @@
import { describe, expect } from "bun:test"
import { Effect, Layer, Stream } from "effect"
import { Effect } from "effect"
import { LLM, LLMEvent, Message } from "../../src/index.js"
import { XAI } from "../../src/providers.js"
import { OpenResponses } from "../../src/protocols/open-responses.js"
import { OpenAIResponses } from "../../src/protocols/openai-responses.js"
import * as ProviderShared from "../../src/protocols/shared.js"
import { XAIResponses } from "../../src/protocols/xai-responses.js"
import {
LLMClient,
RequestExecutor,
WebSocketTransport,
type ChannelCheckpoint,
type WebSocketChannelDriver,
} from "../../src/route.js"
import { LLMClient } from "../../src/route.js"
import { compileRequest } from "../../src/route/client.js"
import { it } from "../lib/effect.js"
import { fixedResponse } from "../lib/http.js"
@@ -20,35 +13,6 @@ import { sseEvents } from "../lib/sse.js"
const model = XAI.configure({ apiKey: "test", baseURL: "https://api.x.ai/v1" }).responses("grok-4.6")
/** Runs a request through the WebSocket transport and hands back its channel driver; the HTTP fallback answers. */
const channelDriver = (request: ReturnType<typeof LLM.request>) =>
Effect.gen(function* () {
let driver: WebSocketChannelDriver | undefined
yield* LLMClient.generate(request, {
webSocket: {
execute: (exchange) =>
Effect.sync(() => {
driver = exchange.driver
return { frames: exchange.fallback(), complete: Effect.void }
}),
},
}).pipe(Effect.provide(fixedResponse(sseEvents({ type: "response.completed", response: { id: "http" } }))))
if (!driver) throw new Error("Expected a WebSocket channel driver")
return driver
})
const completed = (driver: WebSocketChannelDriver, id: string) =>
Effect.gen(function* () {
const create = yield* driver.create(undefined)
yield* driver.observe(create, ProviderShared.encodeJson({ type: "response.created", response: { id } }))
const observation = yield* driver.observe(
create,
ProviderShared.encodeJson({ type: "response.completed", response: { id } }),
)
if (observation.type !== "completed" || !observation.checkpoint) throw new Error("Expected a checkpoint")
return observation.checkpoint
})
describe("xAI Responses route", () => {
it.effect("composes the Open Responses baseline with xAI extensions", () =>
Effect.gen(function* () {
@@ -198,78 +162,6 @@ describe("xAI Responses route", () => {
}),
)
it.effect("classifies xAI's untyped WebSocket error envelope", () =>
Effect.gen(function* () {
// xAI answers a rejected response.create with an error envelope that carries no event type.
const envelope = ProviderShared.encodeJson({
error: {
message:
'Request validation error: {"code":"400","error":"Argument not supported: instructions and previous_response_id together"}',
type: "api_error",
},
})
const webSocket = WebSocketTransport.makeDirect({
open: () =>
Effect.succeed({ sendText: () => Effect.void, messages: Stream.make(envelope), close: Effect.void }),
})
const error = yield* LLMClient.generate(LLM.request({ model, prompt: "Hello" }), { webSocket }).pipe(
Effect.provide(
LLMClient.layer.pipe(
Layer.provide(
Layer.succeed(
RequestExecutor.Service,
RequestExecutor.Service.of({ execute: () => Effect.die("unexpected HTTP request") }),
),
),
),
),
Effect.flip,
)
expect(error.reason._tag).toBe("ProviderInternal")
expect(error.message).toContain("Argument not supported: instructions and previous_response_id together")
expect(error.reason.body).toBe(envelope)
}),
)
it.effect("continues stored responses without instructions and sends unstored steps in full", () =>
Effect.gen(function* () {
const step = (store: boolean, ...prompts: string[]) =>
LLM.request({
model,
system: "You are terse.",
messages: prompts.map((prompt) => Message.user(prompt)),
providerOptions: { store },
})
const send = (store: boolean, checkpoint: ChannelCheckpoint) =>
channelDriver(step(store, "First", "Second")).pipe(Effect.flatMap((driver) => driver.create(checkpoint)))
const stored = yield* send(true, yield* completed(yield* channelDriver(step(true, "First")), "resp_1"))
expect(stored.mode).toBe("incremental")
expect(JSON.parse(stored.message)).toEqual({
type: "response.create",
model: "grok-4.6",
store: true,
include: ["reasoning.encrypted_content"],
previous_response_id: "resp_1",
input: [{ role: "user", content: [{ type: "input_text", text: "Second" }] }],
})
// The connection cache only serves stored responses, so the default store: false never chains.
const unstored = yield* send(false, yield* completed(yield* channelDriver(step(false, "First")), "resp_1"))
expect(unstored.mode).toBe("full")
expect(JSON.parse(unstored.message)).toMatchObject({
instructions: "You are terse.",
store: false,
input: [
{ role: "user", content: [{ type: "input_text", text: "First" }] },
{ role: "user", content: [{ type: "input_text", text: "Second" }] },
],
})
expect(JSON.parse(unstored.message).previous_response_id).toBeUndefined()
}),
)
it.effect("parses xAI hosted tool items", () =>
Effect.gen(function* () {
const item = { type: "x_search_call", id: "x_search_1", status: "completed", action: { query: "news" } }
@@ -1,103 +0,0 @@
import { DialogProvider } from "@opencode/ui/context/dialog"
import { Browser } from "@opencode/plugin-browser/rpc"
import { For, Show } from "solid-js"
import { createStore } from "solid-js/store"
import { render } from "solid-js/web"
import { LanguageProvider, UiI18nBridge } from "../src/runtime/i18n/language"
import type { BrowserPaneLayout, BrowserPaneRegistration } from "../src/runtime/platform/browser-pane"
import type { createSessionBrowser } from "../src/session/browser/model"
import { SessionBrowserPane } from "../src/session/browser/pane"
export function mountBrowserPane() {
const host = document.createElement("main")
host.dataset.testid = "browser-pane-fixture"
host.style.cssText = "position:fixed;inset:0;z-index:1000;background:#181818;color:#eee;padding:24px"
document.body.appendChild(host)
function Fixture() {
const [store, setStore] = createStore({
session: "Alpha",
mounted: true,
visible: true,
layouts: {} as Record<string, BrowserPaneLayout | undefined>,
})
const tabs = ["Alpha", "Beta"].map((name) => ({
id: Browser.TabID.make(`tab_${name === "Alpha" ? "11111111" : "22222222"}-1111-1111-1111-111111111111`),
title: name,
url: `https://${name.toLowerCase()}.example/`,
loading: false,
canGoBack: false,
canGoForward: false,
generation: 0,
}))
// Record the native boundary per registration: hiding Beta cannot hide Alpha's view.
const registrations = new Map<string, BrowserPaneRegistration>(
tabs.map((tab) => [
tab.title,
{
setLayout: (layout) => setStore("layouts", tab.title, layout),
command: async () => undefined,
close: () => undefined,
},
]),
)
const browser: ReturnType<typeof createSessionBrowser> = {
available: () => true,
attached: () => !!registrations.get(store.session),
opened: () => !!registrations.get(store.session),
state: () => ({ tabs: tabs.filter((tab) => tab.title === store.session), focusedTabID: null }),
tabs: () => tabs.filter((tab) => tab.title === store.session),
active: () => tabs.find((tab) => tab.title === store.session) ?? tabs[0],
registration: () => registrations.get(store.session),
error: () => undefined,
suspended: () => false,
close: () => undefined,
open: () => undefined,
command: () => undefined,
}
return (
<>
<h1 style={{ "font-size": "24px", "margin-bottom": "16px" }}>Browser pane lifecycle</h1>
<p>Selected session: {store.session}</p>
<nav style={{ display: "flex", gap: "20px", margin: "16px 0" }}>
<For each={["Alpha", "Beta", "Empty"]}>
{(name) => <button onClick={() => setStore({ session: name, mounted: name !== "Empty" })}>{name}</button>}
</For>
<button onClick={() => setStore("mounted", false)}>Unmount pane</button>
<button onClick={() => setStore("visible", (visible) => !visible)}>Toggle Review tab</button>
</nav>
<div style={{ width: "640px", height: "360px", border: "1px solid #555" }}>
<Show when={store.mounted}>
<SessionBrowserPane browser={browser} visible={store.visible} />
</Show>
</div>
<h2 style={{ "font-size": "18px", margin: "20px 0 12px" }}>Native layout recorder</h2>
<p>The desktop boundary keeps each session's page visible until its registration is hidden.</p>
<For each={tabs}>
{(tab) => (
<div
data-testid={`native-${tab.title}`}
data-visible={!!store.layouts[tab.title]?.visible}
style={{ padding: "12px", margin: "8px 0", border: "1px solid #555" }}
>
{tab.title}: {store.layouts[tab.title]?.visible ? "visible" : "hidden"}
</div>
)}
</For>
</>
)
}
return render(
() => (
<LanguageProvider locale="en">
<UiI18nBridge>
<DialogProvider>
<Fixture />
</DialogProvider>
</UiI18nBridge>
</LanguageProvider>
),
host,
)
}
@@ -1,44 +0,0 @@
import { fileURLToPath } from "node:url"
import { expect, story } from "../../storybook/playwright/story"
const fixture = `/@fs/${fileURLToPath(new URL("./browser-pane.fixture.tsx", import.meta.url)).replaceAll("\\", "/")}`
story.beforeEach(async ({ mount, page }) => {
const component = await mount("opencode-composer-flow--mixed-attachments")
await expect(component.getByRole("textbox", { name: "Prompt", exact: true })).toBeVisible()
await page.evaluate(async (fixture) => {
const { mountBrowserPane } = await import(fixture)
mountBrowserPane()
}, fixture)
await expect(page.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "true")
})
story("hides the previous registration when the mounted pane switches sessions", async ({ page }, testInfo) => {
const root = page.getByTestId("browser-pane-fixture")
await root.getByRole("button", { name: "Beta", exact: true }).click()
await expect(root.getByTestId("native-Beta")).toHaveAttribute("data-visible", "true")
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "false")
await page.screenshot({ path: testInfo.outputPath("session-switch.png") })
await root.getByRole("button", { name: "Alpha", exact: true }).click()
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "true")
await expect(root.getByTestId("native-Beta")).toHaveAttribute("data-visible", "false")
})
story("hides the outgoing browser when the destination has no browser pane", async ({ page }) => {
const root = page.getByTestId("browser-pane-fixture")
await root.getByRole("button", { name: "Empty", exact: true }).click()
await expect(root.locator("#browser-panel")).toHaveCount(0)
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "false")
await root.getByRole("button", { name: "Alpha", exact: true }).click()
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "true")
})
story("hides and restores the same registration for Review tabs and unmount", async ({ page }) => {
const root = page.getByTestId("browser-pane-fixture")
await root.getByRole("button", { name: "Toggle Review tab", exact: true }).click()
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "false")
await root.getByRole("button", { name: "Toggle Review tab", exact: true }).click()
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "true")
await root.getByRole("button", { name: "Unmount pane", exact: true }).click()
await expect(root.getByTestId("native-Alpha")).toHaveAttribute("data-visible", "false")
})
@@ -1,80 +0,0 @@
import { expect, story } from "../../storybook/playwright/story"
story("settings menu reconnect retains its prompt handler across server updates", async ({ mount, page }) => {
const component = await mount("app-dialog-ssh--settings-reconnect")
await component.getByRole("button", { name: "More options" }).click()
await page.getByRole("menuitem", { name: "Connect", exact: true }).click()
const dialog = page.getByRole("dialog")
await expect(dialog.getByRole("textbox")).toBeVisible()
await dialog.getByRole("textbox").fill("fixture-password")
await dialog.getByRole("button", { name: "Continue" }).click()
await expect(dialog.getByRole("textbox", { name: "Verification code:" })).toBeVisible()
await dialog.getByRole("textbox").fill("123456")
await dialog.getByRole("button", { name: "Continue" }).click()
await expect(dialog).toHaveCount(0)
await expect(component.getByRole("button", { name: "Authenticate", exact: true })).toHaveCount(0)
})
story("cancelling a version mismatch permits reconnecting again", async ({ mount, page }) => {
const component = await mount("app-dialog-ssh--incompatible-session")
await component.getByRole("button", { name: "Reconnect", exact: true }).click()
const dialog = page.getByRole("dialog")
await expect(dialog.getByRole("status")).toContainText("Server update required")
await expect(dialog.getByRole("textbox")).toHaveCount(0)
await dialog.getByRole("button", { name: "Cancel", exact: true }).click()
await expect(dialog).toHaveCount(0)
await component.getByRole("button", { name: "Reconnect", exact: true }).click()
await expect(dialog.getByRole("status")).toContainText("Server update required")
await expect(dialog.getByRole("textbox")).toHaveCount(0)
await dialog.getByRole("button", { name: "Cancel", exact: true }).click()
await expect(dialog).toHaveCount(0)
})
story("adding a server keeps all SSH challenges in the original connection dialog", async ({ mount, page }) => {
await mount("app-dialog-ssh--host")
const dialog = page.getByRole("dialog")
await dialog.getByRole("textbox", { name: "Host or SSH command" }).fill("ssh devbox")
await dialog.getByRole("button", { name: "Add server", exact: true }).click()
await expect(dialog).toHaveCount(1)
await expect(dialog.getByRole("button", { name: "Cancel", exact: true })).toBeFocused()
await dialog.getByRole("button", { name: "Trust and connect" }).click()
await dialog.getByRole("textbox").fill("fixture-password")
await dialog.getByRole("button", { name: "Continue", exact: true }).click()
await dialog.getByRole("textbox", { name: "Verification code:" }).fill("123456")
await dialog.getByRole("button", { name: "Continue", exact: true }).click()
await expect(dialog).toHaveCount(0)
})
story("adding an incompatible server advances to a dedicated update step", async ({ mount, page }) => {
await mount("app-dialog-ssh--incompatible-host")
const dialog = page.getByRole("dialog")
await dialog.getByRole("textbox", { name: "Host or SSH command" }).fill("ssh devbox")
await dialog.getByRole("button", { name: "Add server", exact: true }).click()
await expect(dialog.getByRole("status")).toContainText("Server update required")
await expect(dialog.getByRole("textbox")).toHaveCount(0)
await expect(dialog.getByRole("alert")).toHaveCount(0)
await expect(dialog.getByRole("button", { name: "Update and reconnect", exact: true })).toBeVisible()
})
story("updating an incompatible connection continues authentication in the same dialog", async ({ mount, page }) => {
const component = await mount("app-dialog-ssh--incompatible-session")
await component.getByRole("button", { name: "Reconnect", exact: true }).click()
const dialog = page.getByRole("dialog")
await dialog.getByRole("button", { name: "Update and reconnect", exact: true }).click()
await expect(dialog).toHaveCount(1)
await dialog.getByRole("textbox").fill("fixture-password")
await dialog.getByRole("button", { name: "Continue", exact: true }).click()
await dialog.getByRole("textbox", { name: "Verification code:" }).fill("123456")
await dialog.getByRole("button", { name: "Continue", exact: true }).click()
await expect(dialog).toHaveCount(0)
await expect(component.getByText("Session connected")).toBeVisible()
})
story("key-based reconnect completes without opening an authentication dialog", async ({ mount, page }) => {
const component = await mount("app-dialog-ssh--key-reconnect")
await component.getByRole("button", { name: "Reconnect", exact: true }).click()
await expect(component.getByRole("button", { name: "Connecting to SSH server" })).toBeDisabled()
await expect(page.getByRole("dialog")).toHaveCount(0)
await expect(component.getByText("Session connected")).toBeVisible()
await expect(page.getByRole("dialog")).toHaveCount(0)
})
@@ -1,27 +0,0 @@
# Session-export load benchmark
Replay an exported session against a production app build. The two cases compare the default Compact preset with every category ungrouped and details still collapsed.
From `packages/app` in PowerShell:
```powershell
$env:PLAYWRIGHT_BUILD = '1'
$env:PLAYWRIGHT_BASE_URL = 'http://127.0.0.1:4398' # Existing production preview
$env:LAGGY_SESSION_FILE = 'C:\path\session.json'
$env:LAGGY_SESSION_OUTPUT = 'C:\tmp\opencode\session-load'
$env:LAGGY_SESSION_HISTORY = 'paged' # Or 'full' to supply all exported history
bun x playwright test --config e2e/performance/playwright.config.ts timeline/laggy-session-benchmark.spec.ts --repeat-each=20 --workers=1 --retries=0
bun e2e/performance/timeline/laggy-session-report.ts $env:LAGGY_SESSION_OUTPUT
```
Each test uses a fresh browser context and measures one cold load, switches back to the source session, then measures one warm load. Repetitions therefore interleave `cold → warm` pairs rather than collecting separate cold and warm batches. There are no discarded warm-up switches. The warm member of every pair must issue zero message requests. Each pair is saved in a separate JSON file; compare paired differences as well as the cold and warm distributions when system load varies.
The report writes `summary.json` and prints the median, p95, maximum, and median paired cold-minus-warm difference. It rejects incomplete or cold-only records instead of mixing them into paired results.
`--repeat-each=20` collects 20 pairs per grouping mode. `LAGGY_SESSION_COLD_ONLY=1` remains available for focused cold profiling. Screenshots are taken after a pair finishes, not between its measurements.
The app shell, source session, model control, and fonts are ready before the timed action. These measurements cover session entry, not application startup. `firstCorrectObservedMs` begins at mousedown and ends when the destination is visible at its expected bottom position, including Compact's automatic history fill. `stableObservedMs` includes three-observation confirmation and must not be treated as additional rendering time.
Set `OPENCODE_PERFORMANCE_TRACE_DIR` for Chrome traces. `LAGGY_TRACE_ITERATION=0` traces the cold load; the default (`1`) traces the warm member of the pair. Profile separately from timing runs.
`LAGGY_HTTP=1` disables route interception for an external HTTP replay server containing the same export and source fixture. Keep direct HTTP and Playwright-routed cold series separate: routing adds transport overhead. Raw samples, mode settings, viewport, browser version, and screenshots are retained in the output directory. The export itself is not copied into the repository.
@@ -1,227 +0,0 @@
import { readFileSync, mkdirSync, writeFileSync } from "node:fs"
import type { SessionMessageInfo } from "@opencode/client/promise"
import { base64Encode } from "@opencode/util/encode"
import { timelineCategories, timelinePresets } from "@opencode/session-ui/timeline/detail"
import { mockOpenCodeServer } from "../../utils/mock-server"
import { expectSessionTitle } from "../../utils/waits"
import { benchmark, expect } from "../benchmark"
import { measureSessionSwitch, waitForStableTimeline } from "./session-tab-switch-probe"
import { stressSessionHref } from "./timeline-test-helpers"
import { startChromeTrace } from "../chrome-trace"
const file = process.env.LAGGY_SESSION_FILE
const session = file
? (JSON.parse(readFileSync(file, "utf8")) as {
info: {
id: string
projectID: string
title: string
model?: { id: string; providerID: string }
location: { directory: string }
time: { created: number; updated: number }
}
messages: SessionMessageInfo[]
})
: undefined
const sourceID = "ses_laggy_benchmark_source"
const sourceMessageID = "msg_laggy_benchmark_source"
const history = process.env.LAGGY_SESSION_HISTORY ?? "full"
const viewport = { width: 1440, height: 900 }
benchmark.use({ viewport, video: "off", trace: "off", serviceWorkers: "block", traceScope: "interaction" })
for (const mode of ["compact", "ungrouped"] as const) {
benchmark(`laggy session: ${mode}`, async ({ page, report }, testInfo) => {
benchmark.skip(!session, "Set LAGGY_SESSION_FILE to a session export")
if (!session) return
const output = process.env.LAGGY_SESSION_OUTPUT ?? testInfo.outputPath("session-load")
const model = session.info.model ?? { id: "benchmark-model", providerID: "benchmark" }
const lastID = session.messages.findLast((message) => message.type === "user")!.id
const lastText = session.messages.findLast(
(message) =>
message.type === "assistant" && message.content.some((part) => part.type === "text" && part.text.trim()),
)!
benchmark.setTimeout(Number(process.env.LAGGY_SESSION_TIMEOUT ?? 180_000))
const requests: string[] = []
const errors: string[] = []
page.on("pageerror", (error) => errors.push(error.message))
if (process.env.LAGGY_HTTP === "1")
page.on("request", (request) => {
const match = new URL(request.url()).pathname.match(/^\/api\/session\/([^/]+)\/message$/)
if (request.method() === "GET" && match) requests.push(decodeURIComponent(match[1]))
})
const detail = Object.fromEntries(
timelineCategories.map((category) => [
category,
{
...timelinePresets[2].value[category],
placement: mode === "compact" ? "grouped" : "separate",
},
]),
)
const directory = session.info.location.directory
if (process.env.LAGGY_HTTP !== "1")
await mockOpenCodeServer(page, {
directory,
project: {
id: session.info.projectID,
worktree: directory,
vcs: "git",
name: "session-benchmark",
time: session.info.time,
sandboxes: [],
},
provider: {
all: [
{
id: model.providerID,
name: model.providerID,
models: { [model.id]: { id: model.id, name: model.id, limit: { context: 1_000_000 } } },
},
],
connected: [model.providerID],
default: { providerID: model.providerID, modelID: model.id },
},
sessions: [session.info, { ...session.info, id: sourceID, title: "Benchmark source" }],
pageMessages: (id, limit, before) => {
if (id !== session.info.id)
return {
items: [
{
id: sourceMessageID,
type: "user",
text: "Benchmark source",
time: { created: session.info.time.created },
},
],
}
if (history === "full") return { items: session.messages }
const end = before ? session.messages.findIndex((message) => message.id === before) : session.messages.length
const start = Math.max(0, end - limit)
return {
items: session.messages.slice(start, end),
cursor: start > 0 ? session.messages[start].id : undefined,
}
},
onMessages: (request) => {
if (request.phase === "start") requests.push(request.sessionID)
},
})
await page.addInitScript(
({ detail, directory, server, sessionIDs, dirBase64 }) => {
localStorage.setItem("settings.v3", JSON.stringify({ general: { timelineDetail: detail } }))
localStorage.setItem(
"opencode.global.dat:server",
JSON.stringify({
projects: { local: [{ worktree: directory, expanded: true }] },
lastProject: { local: directory },
}),
)
localStorage.setItem(
"opencode.window.browser.dat:tabs",
JSON.stringify(sessionIDs.map((sessionId) => ({ type: "session", server, dirBase64, sessionId }))),
)
},
{
detail,
directory,
server: process.env.PLAYWRIGHT_BASE_URL!,
sessionIDs: [sourceID, session.info.id],
dirBase64: base64Encode(directory),
},
)
await page.goto(stressSessionHref(sourceID))
await expectSessionTitle(page, "Benchmark source")
await expect(page.locator('[data-slot="user-message-text"]')).toHaveText("Benchmark source")
await expect(page.getByRole("textbox", { name: "Prompt", exact: true })).toBeEditable()
await expect(page.getByRole("button", { name: model.id, exact: true })).toBeVisible()
await page.evaluate(() => document.fonts.ready.then(() => undefined))
expect(requests).toEqual([sourceID])
const startedAt = new Date().toISOString()
const samples = []
const phases = process.env.LAGGY_SESSION_COLD_ONLY === "1" ? (["cold"] as const) : (["cold", "warm"] as const)
for (const [iteration, phase] of phases.entries()) {
const before = requests.length
const stopTrace =
iteration === Number(process.env.LAGGY_TRACE_ITERATION ?? 1)
? await startChromeTrace(page, `laggy-${history}-${mode}`)
: undefined
const result = await measureSessionSwitch(page, {
destinationIDs: session.messages.map((message) => message.id),
sourceIDs: [sourceMessageID],
lastID,
requiredPartID: history === "paged" && mode === "compact" ? `${lastText.id}:text:0` : undefined,
requireBottomAnchor: true,
href: stressSessionHref(session.info.id),
switch: async () => {
await page.locator(`[data-slot="titlebar-tabs"] a[href="${stressSessionHref(session.info.id)}"]`).click()
},
})
await stopTrace?.()
await expectSessionTitle(page, session.info.title)
if (history === "full" || mode === "ungrouped") await waitForStableTimeline(page, lastID)
await expect(
page.locator('[data-timeline-key] [data-component="markdown"]:not([data-markdown-ready])'),
).toHaveCount(0)
expect(result.firstCorrectObservedMs).not.toBeNull()
expect(result.stableObservedMs).not.toBeNull()
if (phase === "warm") expect(requests.length - before).toBe(0)
samples.push({
iteration,
phase,
messageRequests: requests.length - before,
messageResources: await page.evaluate((sessionID) => {
const start = performance.getEntriesByName("session-switch:start").at(-1)!.startTime
return (performance.getEntriesByType("resource") as PerformanceResourceTiming[])
.filter(
(entry) =>
entry.startTime >= start && new URL(entry.name).pathname === `/api/session/${sessionID}/message`,
)
.map((entry) => ({
limit: Number(new URL(entry.name).searchParams.get("limit")),
startMs: entry.startTime - start,
durationMs: entry.duration,
transferBytes: entry.transferSize,
}))
}, session.info.id),
...result,
})
if (iteration === phases.length - 1) {
mkdirSync(output, { recursive: true })
if (testInfo.repeatEachIndex === 0) await page.screenshot({ path: `${output}/${mode}.png` })
break
}
await page.locator(`[data-slot="titlebar-tabs"] a[href="${stressSessionHref(sourceID)}"]`).click()
await expectSessionTitle(page, "Benchmark source")
await expect(page.locator('[data-slot="user-message-text"]')).toHaveText("Benchmark source")
await expect(page.getByRole("textbox", { name: "Prompt", exact: true })).toBeEditable()
}
expect(errors).toEqual([])
const result = {
pair: testInfo.repeatEachIndex,
startedAt,
mode,
history,
file,
messages: session.messages.length,
viewport,
browser: page.context().browser()!.version(),
detail,
samples,
}
writeFileSync(`${output}/${mode}-${testInfo.repeatEachIndex}.json`, JSON.stringify(result, null, 2))
report(
{ samples },
{
mode,
sampling: "cold/warm pair in one browser context",
pair: testInfo.repeatEachIndex,
messages: session.messages.length,
viewport,
data: history === "full" ? "full exported history" : "paginated exported history",
transport: process.env.LAGGY_HTTP === "1" ? "http" : "playwright-route",
inputEvent: "mousedown",
},
)
})
}
@@ -1,63 +0,0 @@
export {}
type Pair = {
mode: "compact" | "ungrouped"
samples: { phase: string; firstCorrectObservedMs: number | null; messageRequests: number }[]
}
const directory = Bun.argv[2]
if (!directory) throw new Error("Pass the directory containing session-load pairs")
const pairs = await Promise.all(
[...new Bun.Glob("{compact,ungrouped}-*.json").scanSync(directory)].map(async (file) => {
const pair = (await Bun.file(`${directory}/${file}`).json()) as Pair
const cold = pair.samples.find((sample) => sample.phase === "cold")
const warm = pair.samples.find((sample) => sample.phase === "warm")
if (cold?.firstCorrectObservedMs == null || warm?.firstCorrectObservedMs == null)
throw new Error(`Expected a completed cold/warm pair in ${file}`)
return {
mode: pair.mode,
cold: cold.firstCorrectObservedMs,
warm: warm.firstCorrectObservedMs,
requests: warm.messageRequests,
}
}),
)
if (!pairs.length) throw new Error(`No session-load pairs found in ${directory}`)
const result = ["compact", "ungrouped"].flatMap((mode) => {
const selected = pairs.filter((pair) => pair.mode === mode)
if (!selected.length) return []
return [
{
mode,
cold: { ...stats(selected.map((pair) => pair.cold)), over50ms: selected.filter((pair) => pair.cold > 50).length },
warm: { ...stats(selected.map((pair) => pair.warm)), over50ms: selected.filter((pair) => pair.warm > 50).length },
pairedColdMinusWarm: stats(selected.map((pair) => pair.cold - pair.warm)),
messageRequestsDuringWarm: selected.reduce((total, pair) => total + pair.requests, 0),
},
]
})
await Bun.write(`${directory}/summary.json`, JSON.stringify(result, null, 2))
console.table(
result.map((row) => ({
mode: row.mode,
pairs: row.cold.n,
coldMedianMs: Math.round(row.cold.median * 10) / 10,
coldP95Ms: Math.round(row.cold.p95 * 10) / 10,
warmMedianMs: Math.round(row.warm.median * 10) / 10,
warmP95Ms: Math.round(row.warm.p95 * 10) / 10,
warmMaxMs: Math.round(row.warm.max * 10) / 10,
pairedDifferenceMs: Math.round(row.pairedColdMinusWarm.median * 10) / 10,
})),
)
function stats(values: number[]) {
const sorted = values.toSorted((left, right) => left - right)
return {
n: sorted.length,
median: (sorted[Math.floor((sorted.length - 1) / 2)] + sorted[Math.floor(sorted.length / 2)]) / 2,
p95: sorted[Math.ceil(sorted.length * 0.95) - 1],
min: sorted[0],
max: sorted.at(-1)!,
}
}

Some files were not shown because too many files have changed in this diff Show More