mirror of
https://github.com/anomalyco/opencode.git
synced 2026-09-23 17:17:43 +00:00
Compare commits
175
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ca8bcf5b47 | ||
|
|
b3b030856b | ||
|
|
a117ebb408 | ||
|
|
f2d9ebb886 | ||
|
|
0332a26be6 | ||
|
|
bb7876dfa8 | ||
|
|
dbaa57a21b | ||
|
|
b6bc55764c | ||
|
|
14a3311a61 | ||
|
|
dddb5eb96f | ||
|
|
affa57e40f | ||
|
|
681ea07a92 | ||
|
|
6e89d9e2c3 | ||
|
|
f526727178 | ||
|
|
9c8a63e852 | ||
|
|
fabf56781c | ||
|
|
a25d304201 | ||
|
|
cc8886c8bb | ||
|
|
8ce629be22 | ||
|
|
68b28bdb98 | ||
|
|
150dc69e4b | ||
|
|
d5d4461e67 | ||
|
|
d56ce74373 | ||
|
|
17abc5906b | ||
|
|
8656838a5b | ||
|
|
53179daefa | ||
|
|
bab26d63ea | ||
|
|
f0381e5da3 | ||
|
|
740072694d | ||
|
|
2e4abeb25d | ||
|
|
43f1dad8e1 | ||
|
|
cf4b4c2312 | ||
|
|
3bf8a5a8cf | ||
|
|
ddeb19790a | ||
|
|
fe0d1682ca | ||
|
|
1746672c42 | ||
|
|
3a2203eaac | ||
|
|
38c320ea4c | ||
|
|
5c53cfc342 | ||
|
|
8683406690 | ||
|
|
126294a322 | ||
|
|
eea247598b | ||
|
|
d2bbefbac8 | ||
|
|
ad756ef09b | ||
|
|
94df7a812d | ||
|
|
7af65eff37 | ||
|
|
bf6788b94c | ||
|
|
54fbf6d14d | ||
|
|
cdccde7408 | ||
|
|
51d2b66760 | ||
|
|
60c78ed8ab | ||
|
|
f2bdee6726 | ||
|
|
3584eca0eb | ||
|
|
ad1a4a6539 | ||
|
|
10aa949f43 | ||
|
|
067a528b1d | ||
|
|
788f0affcb | ||
|
|
18eeb3201d | ||
|
|
8864eb507e | ||
|
|
e0ddc47aa4 | ||
|
|
2f06f9d58b | ||
|
|
8e62ad7adc | ||
|
|
956de96d8b | ||
|
|
be4e5a6d06 | ||
|
|
fbacf6a126 | ||
|
|
9c18abce47 | ||
|
|
080b7671de | ||
|
|
dcfe1ec7bd | ||
|
|
ceace24a3e | ||
|
|
19e1357a06 | ||
|
|
4b381ac6a1 | ||
|
|
07d48e1ffb | ||
|
|
9fdcb8da41 | ||
|
|
ba61ac6730 | ||
|
|
94b9133910 | ||
|
|
6f8c5ae0aa | ||
|
|
60673aaef3 | ||
|
|
651529d64e | ||
|
|
532f25d0d4 | ||
|
|
643c4c3500 | ||
|
|
9d531435b4 | ||
|
|
5b9dc35eec | ||
|
|
1814dd9799 | ||
|
|
1e1cd042ea | ||
|
|
02566f6219 | ||
|
|
f488aa3f79 | ||
|
|
97457ec7a3 | ||
|
|
d62049aab4 | ||
|
|
096ac95773 | ||
|
|
4cc9b90f27 | ||
|
|
990463aa9f | ||
|
|
cd39063622 | ||
|
|
4c944a86d7 | ||
|
|
97e833a297 | ||
|
|
b8aa08f260 | ||
|
|
3f0118022b | ||
|
|
6f76c31ca7 | ||
|
|
f90beeb9b8 | ||
|
|
4b9a3d80fc | ||
|
|
1316576720 | ||
|
|
2f0c861af0 | ||
|
|
4d94777d4d | ||
|
|
46ebde65e9 | ||
|
|
932c12ad1d | ||
|
|
6f655dcbab | ||
|
|
ab60f08c69 | ||
|
|
60ed84ecd1 | ||
|
|
8aebed170a | ||
|
|
14148a0ea4 | ||
|
|
7f51fbd878 | ||
|
|
cbdd1f66da | ||
|
|
788eb0fa29 | ||
|
|
58fcad77a8 | ||
|
|
c555559ac1 | ||
|
|
e3a3fa7108 | ||
|
|
6238af397e | ||
|
|
7ffd75faf6 | ||
|
|
1f8ab95695 | ||
|
|
62dc1f7696 | ||
|
|
66f10ab7bf | ||
|
|
1ca8f63a79 | ||
|
|
03bcdd580d | ||
|
|
cc502f7e5f | ||
|
|
64ce8771c0 | ||
|
|
702a73d91a | ||
|
|
991b727eb8 | ||
|
|
ba342ce227 | ||
|
|
1464545665 | ||
|
|
fecacc9e68 | ||
|
|
1f73b4806b | ||
|
|
da2ce02596 | ||
|
|
25f35dcfb8 | ||
|
|
717f81ce08 | ||
|
|
5fdfcc7a80 | ||
|
|
0530c8e512 | ||
|
|
1d8cf4564b | ||
|
|
7e88f6bb18 | ||
|
|
af592fb779 | ||
|
|
7fc3f68007 | ||
|
|
cdcbb0047e | ||
|
|
a1956a7522 | ||
|
|
55bc7fd403 | ||
|
|
3049b1e684 | ||
|
|
1f36a7aff8 | ||
|
|
eb0e26b974 | ||
|
|
6f2b0e7833 | ||
|
|
f153255942 | ||
|
|
fef2fad76f | ||
|
|
f30d06ea34 | ||
|
|
dfa44e94e8 | ||
|
|
b81e10a461 | ||
|
|
65c93b69ed | ||
|
|
b073b052d3 | ||
|
|
728b2b6052 | ||
|
|
cc3ce20ab4 | ||
|
|
4ad5001be2 | ||
|
|
14b3c2ea4b | ||
|
|
558bd54c9b | ||
|
|
7e70f7e1ab | ||
|
|
cb6d95b7ef | ||
|
|
7020944359 | ||
|
|
4b00dd2713 | ||
|
|
b447627f5f | ||
|
|
5848ee0d24 | ||
|
|
839aa25c3e | ||
|
|
c42f1c9232 | ||
|
|
417f6d234d | ||
|
|
45a13af0ee | ||
|
|
e50d845451 | ||
|
|
56816621f8 | ||
|
|
7125f5f8b5 | ||
|
|
01a6ed8d97 | ||
|
|
004583d598 | ||
|
|
cbd911368e | ||
|
|
8bfb247854 |
@@ -53,8 +53,6 @@ runs:
|
||||
with:
|
||||
path: ${{ steps.cache.outputs.dir }}
|
||||
key: ${{ runner.os }}-bun-${{ hashFiles('**/bun.lock') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-bun-
|
||||
|
||||
- name: Install setuptools for distutils compatibility
|
||||
run: python3 -m pip install setuptools || pip install setuptools || true
|
||||
@@ -66,9 +64,9 @@ runs:
|
||||
# e.g. ./patches/ for standard-openapi
|
||||
# https://github.com/oven-sh/bun/issues/28147
|
||||
if [ "$RUNNER_OS" = "Windows" ]; then
|
||||
bun install --linker hoisted ${{ inputs.install-flags }}
|
||||
bun install --frozen-lockfile --linker hoisted ${{ inputs.install-flags }}
|
||||
else
|
||||
bun install ${{ inputs.install-flags }}
|
||||
bun install --frozen-lockfile ${{ inputs.install-flags }}
|
||||
fi
|
||||
shell: bash
|
||||
|
||||
|
||||
@@ -112,11 +112,12 @@ jobs:
|
||||
- name: Run unit tests
|
||||
timeout-minutes: 20
|
||||
run: |
|
||||
# The runners have four vCPUs, and each Bun test process performs its own concurrent work.
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
GITHUB_ACTIONS=false bun turbo test
|
||||
GITHUB_ACTIONS=false bun turbo test --concurrency=3
|
||||
exit 0
|
||||
fi
|
||||
GITHUB_ACTIONS=false bun turbo test --affected
|
||||
GITHUB_ACTIONS=false bun turbo test --affected --concurrency=3
|
||||
env:
|
||||
OPENCODE_EXPERIMENTAL_DISABLE_FILEWATCHER: ${{ runner.os == 'Windows' && 'true' || 'false' }}
|
||||
TURBO_SCM_BASE: ${{ github.event_name == 'pull_request' && format('{0}^1', github.sha) || github.event.before }}
|
||||
|
||||
@@ -32,7 +32,7 @@
|
||||
},
|
||||
"packages/ai": {
|
||||
"name": "@opencode/ai",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@aws-sdk/credential-providers": "3.1057.0",
|
||||
"@opencode/schema": "workspace:*",
|
||||
@@ -54,7 +54,7 @@
|
||||
},
|
||||
"packages/app": {
|
||||
"name": "@opencode/app",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@corvu/drawer": "catalog:",
|
||||
"@dnd-kit/abstract": "0.5.0",
|
||||
@@ -92,6 +92,7 @@
|
||||
"solid-js": "catalog:",
|
||||
"solid-presence": "0.2.0",
|
||||
"tailwindcss": "4.3.3",
|
||||
"uqr": "0.1.3",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@happy-dom/global-registrator": "20.0.11",
|
||||
@@ -111,8 +112,9 @@
|
||||
},
|
||||
"packages/cli": {
|
||||
"name": "@opencode/cli",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"bin": {
|
||||
"opencode": "./bin/opencode.cjs",
|
||||
"opencode2": "./bin/opencode2.cjs",
|
||||
},
|
||||
"dependencies": {
|
||||
@@ -175,7 +177,7 @@
|
||||
},
|
||||
"packages/client": {
|
||||
"name": "@opencode/client",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/protocol": "workspace:*",
|
||||
"@opencode/schema": "workspace:*",
|
||||
@@ -201,11 +203,10 @@
|
||||
},
|
||||
"packages/codemode": {
|
||||
"name": "@opencode/codemode",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"acorn": "8.15.0",
|
||||
"effect": "catalog:",
|
||||
"typescript": "catalog:",
|
||||
},
|
||||
"devDependencies": {
|
||||
"@tsconfig/bun": "catalog:",
|
||||
@@ -215,7 +216,7 @@
|
||||
},
|
||||
"packages/console/app": {
|
||||
"name": "@opencode/console-app",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@cloudflare/vite-plugin": "1.15.2",
|
||||
"@ibm/plex": "6.4.1",
|
||||
@@ -251,7 +252,7 @@
|
||||
},
|
||||
"packages/console/core": {
|
||||
"name": "@opencode/console-core",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-sts": "3.782.0",
|
||||
"@jsx-email/render": "1.1.1",
|
||||
@@ -278,7 +279,7 @@
|
||||
},
|
||||
"packages/console/function": {
|
||||
"name": "@opencode/console-function",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@openauthjs/openauth": "0.0.0-20250322224806",
|
||||
"@opencode/console-core": "workspace:*",
|
||||
@@ -295,7 +296,7 @@
|
||||
},
|
||||
"packages/console/mail": {
|
||||
"name": "@opencode/console-mail",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@jsx-email/all": "2.2.3",
|
||||
"@jsx-email/cli": "1.4.3",
|
||||
@@ -319,7 +320,7 @@
|
||||
},
|
||||
"packages/console/support": {
|
||||
"name": "@opencode/console-support",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@cloudflare/vite-plugin": "1.15.2",
|
||||
"@opencode/console-core": "workspace:*",
|
||||
@@ -339,7 +340,7 @@
|
||||
},
|
||||
"packages/core": {
|
||||
"name": "@opencode/core",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@ai-sdk/cohere": "3.0.27",
|
||||
"@ai-sdk/gateway": "3.0.104",
|
||||
@@ -367,7 +368,7 @@
|
||||
"drizzle-orm": "catalog:",
|
||||
"effect": "catalog:",
|
||||
"fuzzysort": "3.1.0",
|
||||
"gitlab-ai-provider": "6.12.1",
|
||||
"gitlab-ai-provider": "6.16.0",
|
||||
"google-auth-library": "10.5.0",
|
||||
"gray-matter": "4.0.3",
|
||||
"htmlparser2": "8.0.2",
|
||||
@@ -407,10 +408,10 @@
|
||||
},
|
||||
"packages/desktop": {
|
||||
"name": "@opencode/desktop",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@zip.js/zip.js": "2.7.62",
|
||||
"electron-context-menu": "4.1.2",
|
||||
"electron-context-menu": "5.0.0",
|
||||
"electron-log": "^5",
|
||||
"electron-updater": "6.8.9",
|
||||
"lighthouse": "13.4.1",
|
||||
@@ -437,7 +438,7 @@
|
||||
"drizzle-kit": "catalog:",
|
||||
"drizzle-orm": "catalog:",
|
||||
"effect": "catalog:",
|
||||
"electron": "42.10.1",
|
||||
"electron": "44.4.3",
|
||||
"electron-builder": "26.15.7",
|
||||
"electron-vite": "6.0.0-beta.1",
|
||||
"puppeteer-core": "25.9.0",
|
||||
@@ -456,7 +457,7 @@
|
||||
},
|
||||
"packages/enterprise": {
|
||||
"name": "@opencode/enterprise",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@hono/standard-validator": "catalog:",
|
||||
"@opencode-ai/sdk": "1.18.21",
|
||||
@@ -493,7 +494,7 @@
|
||||
},
|
||||
"packages/function": {
|
||||
"name": "@opencode/function",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@octokit/auth-app": "8.0.1",
|
||||
"@octokit/rest": "catalog:",
|
||||
@@ -509,7 +510,7 @@
|
||||
},
|
||||
"packages/http-recorder": {
|
||||
"name": "@opencode/http-recorder",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@effect/platform-node-shared": "4.0.0-rc.112",
|
||||
},
|
||||
@@ -528,7 +529,7 @@
|
||||
},
|
||||
"packages/httpapi-codegen": {
|
||||
"name": "@opencode/httpapi-codegen",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"effect": "catalog:",
|
||||
"prettier": "3.6.2",
|
||||
@@ -541,7 +542,7 @@
|
||||
},
|
||||
"packages/latex": {
|
||||
"name": "@opencode/latex",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/plugin": "workspace:*",
|
||||
"@opentui/core": "catalog:",
|
||||
@@ -555,7 +556,7 @@
|
||||
},
|
||||
"packages/merman": {
|
||||
"name": "@opencode/merman",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/plugin": "workspace:*",
|
||||
"@opentui/core": "catalog:",
|
||||
@@ -570,7 +571,7 @@
|
||||
},
|
||||
"packages/plugin": {
|
||||
"name": "@opencode/plugin",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@ai-sdk/provider": "3.0.8",
|
||||
"@opencode/ai": "workspace:*",
|
||||
@@ -609,7 +610,7 @@
|
||||
},
|
||||
"packages/plugin-browser": {
|
||||
"name": "@opencode/plugin-browser",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/plugin": "workspace:*",
|
||||
"@opencode/schema": "workspace:*",
|
||||
@@ -639,7 +640,7 @@
|
||||
},
|
||||
"packages/protocol": {
|
||||
"name": "@opencode/protocol",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/schema": "workspace:*",
|
||||
"effect": "catalog:",
|
||||
@@ -654,7 +655,7 @@
|
||||
},
|
||||
"packages/schema": {
|
||||
"name": "@opencode/schema",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@standard-schema/spec": "catalog:",
|
||||
"effect": "catalog:",
|
||||
@@ -678,7 +679,7 @@
|
||||
},
|
||||
"packages/sdk": {
|
||||
"name": "@opencode/sdk",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/client": "workspace:*",
|
||||
"@opencode/core": "workspace:*",
|
||||
@@ -699,7 +700,7 @@
|
||||
},
|
||||
"packages/server": {
|
||||
"name": "@opencode/server",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@effect/platform-node": "catalog:",
|
||||
"@effect/platform-node-shared": "catalog:",
|
||||
@@ -721,7 +722,7 @@
|
||||
},
|
||||
"packages/session-ui": {
|
||||
"name": "@opencode/session-ui",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@kobalte/core": "catalog:",
|
||||
"@opencode/client": "workspace:*",
|
||||
@@ -756,7 +757,7 @@
|
||||
},
|
||||
"packages/simulation": {
|
||||
"name": "@opencode/simulation",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/ai": "workspace:*",
|
||||
"@opencode/core": "workspace:*",
|
||||
@@ -776,7 +777,7 @@
|
||||
},
|
||||
"packages/stats/app": {
|
||||
"name": "@opencode/stats-app",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@ibm/plex": "6.4.1",
|
||||
"@kobalte/core": "catalog:",
|
||||
@@ -810,7 +811,7 @@
|
||||
},
|
||||
"packages/stats/core": {
|
||||
"name": "@opencode/stats-core",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-athena": "3.933.0",
|
||||
"@planetscale/database": "1.19.0",
|
||||
@@ -829,7 +830,7 @@
|
||||
},
|
||||
"packages/stats/server": {
|
||||
"name": "@opencode/stats-server",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-firehose": "3.933.0",
|
||||
"@effect/platform-node": "catalog:",
|
||||
@@ -875,7 +876,7 @@
|
||||
},
|
||||
"packages/theme": {
|
||||
"name": "@opencode/theme",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opentui/core": "catalog:",
|
||||
"effect": "catalog:",
|
||||
@@ -889,7 +890,7 @@
|
||||
},
|
||||
"packages/tui": {
|
||||
"name": "@opencode/tui",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@opencode/client": "workspace:*",
|
||||
"@opencode/core": "workspace:*",
|
||||
@@ -924,7 +925,7 @@
|
||||
},
|
||||
"packages/ui": {
|
||||
"name": "@opencode/ui",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@kobalte/core": "catalog:",
|
||||
"@pierre/diffs": "catalog:",
|
||||
@@ -959,7 +960,7 @@
|
||||
},
|
||||
"packages/util": {
|
||||
"name": "@opencode/util",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@effect/opentelemetry": "catalog:",
|
||||
"@effect/platform-node": "catalog:",
|
||||
@@ -967,10 +968,14 @@
|
||||
"@npmcli/arborist": "catalog:",
|
||||
"@npmcli/config": "10.8.1",
|
||||
"@opentelemetry/api": "1.9.0",
|
||||
"@opentelemetry/context-async-hooks": "2.6.1",
|
||||
"@opentelemetry/exporter-trace-otlp-http": "0.214.0",
|
||||
"@opentelemetry/sdk-trace-base": "2.6.1",
|
||||
"@opentelemetry/sdk-trace-node": "2.6.1",
|
||||
"@opentelemetry/api-logs": "0.219.0",
|
||||
"@opentelemetry/context-async-hooks": "2.8.0",
|
||||
"@opentelemetry/exporter-trace-otlp-http": "0.219.0",
|
||||
"@opentelemetry/resources": "2.8.0",
|
||||
"@opentelemetry/sdk-logs": "0.219.0",
|
||||
"@opentelemetry/sdk-metrics": "2.8.0",
|
||||
"@opentelemetry/sdk-trace-base": "2.8.0",
|
||||
"@opentelemetry/sdk-trace-node": "2.8.0",
|
||||
"cross-spawn": "catalog:",
|
||||
"effect": "catalog:",
|
||||
"glob": "13.0.5",
|
||||
@@ -992,7 +997,7 @@
|
||||
},
|
||||
"packages/web": {
|
||||
"name": "@opencode/web",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"@astrojs/cloudflare": "12.6.3",
|
||||
"@astrojs/markdown-remark": "6.3.1",
|
||||
@@ -1033,7 +1038,7 @@
|
||||
},
|
||||
"services/update": {
|
||||
"name": "@opencode/update",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"dependencies": {
|
||||
"jose": "6.0.11",
|
||||
"semver": "catalog:",
|
||||
@@ -1068,13 +1073,13 @@
|
||||
"trustedDependencies": [
|
||||
"electron",
|
||||
"esbuild",
|
||||
"protobufjs",
|
||||
],
|
||||
"patchedDependencies": {
|
||||
"@pierre/trees@1.0.0-beta.4": "patches/@pierre%2Ftrees@1.0.0-beta.4.patch",
|
||||
"@tanstack/virtual-core@3.17.8": "patches/@tanstack%2Fvirtual-core@3.17.8.patch",
|
||||
"ghostty-web@github:anomalyco/ghostty-web#83c0a07": "patches/ghostty-web@0.3.0.patch",
|
||||
"@modelcontextprotocol/client@2.0.0": "patches/@modelcontextprotocol%2Fclient@2.0.0.patch",
|
||||
"pacote@21.5.1": "patches/pacote@21.5.1.patch",
|
||||
"@standard-community/standard-openapi@0.2.9": "patches/@standard-community%2Fstandard-openapi@0.2.9.patch",
|
||||
"@npmcli/agent@4.0.2": "patches/@npmcli%2Fagent@4.0.2.patch",
|
||||
"@silvia-odwyer/photon-node@0.3.4": "patches/@silvia-odwyer%2Fphoton-node@0.3.4.patch",
|
||||
@@ -2219,31 +2224,31 @@
|
||||
|
||||
"@opentelemetry/api": ["@opentelemetry/api@1.9.0", "", {}, "sha512-3giAOQvZiH5F9bMlMiv8+GSPMeqg0dbaeo58/0SlA9sxSqZhnUtxzX9/2FzyhS9sWQf5S0GJE0AKBrFqjpeYcg=="],
|
||||
|
||||
"@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.214.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-40lSJeqYO8Uz2Yj7u94/SJWE/wONa7rmMKjI1ZcIjgf3MHNHv1OZUCrCETGuaRF62d5pQD1wKIW+L4lmSMTzZA=="],
|
||||
"@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.219.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-FFx7YnaYJlIjqWW/AG/yAZ0L/NEY724PipXXXQLdtZPbLwBGbUMTGL1i/esI56TWfTUXxhLfpgrnWJCG8aUJyg=="],
|
||||
|
||||
"@opentelemetry/context-async-hooks": ["@opentelemetry/context-async-hooks@2.6.1", "", { "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-XHzhwRNkBpeP8Fs/qjGrAf9r9PRv67wkJQ/7ZPaBQQ68DYlTBBx5MF9LvPx7mhuXcDessKK2b+DcxqwpgkcivQ=="],
|
||||
"@opentelemetry/context-async-hooks": ["@opentelemetry/context-async-hooks@2.8.0", "", { "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-/3FIraneMcng67SUJCxvyInk/oxzwsxyadufk0wwfOBLf5wqtAGX4MoQASwSbndBPeARzBryUM9Azr5kHIdWLw=="],
|
||||
|
||||
"@opentelemetry/core": ["@opentelemetry/core@2.6.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-8xHSGWpJP9wBxgBpnqGL0R3PbdWQndL1Qp50qrg71+B28zK5OQmUgcDKLJgzyAAV38t4tOyLMGDD60LneR5W8g=="],
|
||||
"@opentelemetry/core": ["@opentelemetry/core@2.8.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-hd1Lfh8p545nNz+jq1Ejfz+Mn1hyLuxYn1YzTfFNrxr8urEWMNQLPf1Th8kjOH+HxwawCrtgBp8JpBUR4ZSgww=="],
|
||||
|
||||
"@opentelemetry/exporter-trace-otlp-http": ["@opentelemetry/exporter-trace-otlp-http@0.214.0", "", { "dependencies": { "@opentelemetry/core": "2.6.1", "@opentelemetry/otlp-exporter-base": "0.214.0", "@opentelemetry/otlp-transformer": "0.214.0", "@opentelemetry/resources": "2.6.1", "@opentelemetry/sdk-trace-base": "2.6.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-kIN8nTBMgV2hXzV/a20BCFilPZdAIMYYJGSgfMMRm/Xa+07y5hRDS2Vm12A/z8Cdu3Sq++ZvJfElokX2rkgGgw=="],
|
||||
"@opentelemetry/exporter-trace-otlp-http": ["@opentelemetry/exporter-trace-otlp-http@0.219.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/otlp-exporter-base": "0.219.0", "@opentelemetry/otlp-transformer": "0.219.0", "@opentelemetry/resources": "2.8.0", "@opentelemetry/sdk-trace-base": "2.8.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-9t6SvBXXBEjOBcIzgozvBbd3jWrv3Gt3ngGhl1fhdZ/zRc7oZDVOFEqbi2zlBpW9BXhgDMKv422J0DL/3iQWfw=="],
|
||||
|
||||
"@opentelemetry/instrumentation": ["@opentelemetry/instrumentation@0.220.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.220.0", "import-in-the-middle": "^3.0.0", "require-in-the-middle": "^8.0.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-xQx3E2WxP1mDvKzxLxX+CTCtNLa560YJZ3087qYHerl2YmiKpv7AH+dAy7vmx+eVrZ5BwhfWUAVoKOoxCNHcpw=="],
|
||||
|
||||
"@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.214.0", "", { "dependencies": { "@opentelemetry/core": "2.6.1", "@opentelemetry/otlp-transformer": "0.214.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-u1Gdv0/E9wP+apqWf7Wv2npXmgJtxsW2XL0TEv9FZloTZRuMBKmu8cYVXwS4Hm3q/f/3FuCnPTgiwYvIqRSpRg=="],
|
||||
"@opentelemetry/otlp-exporter-base": ["@opentelemetry/otlp-exporter-base@0.219.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/otlp-transformer": "0.219.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-zvIxQX/AZUVKDU+hCuYx+7UkiP7GRdnk1ZbFQRYzHvYp47cAWR4j3IhoPhV9KaeXEv2xdGq3IA6PnpzDmLcmSA=="],
|
||||
|
||||
"@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.214.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.214.0", "@opentelemetry/core": "2.6.1", "@opentelemetry/resources": "2.6.1", "@opentelemetry/sdk-logs": "0.214.0", "@opentelemetry/sdk-metrics": "2.6.1", "@opentelemetry/sdk-trace-base": "2.6.1", "protobufjs": "^7.0.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-DSaYcuBRh6uozfsWN3R8HsN0yDhCuWP7tOFdkUOVaWD1KVJg8m4qiLUsg/tNhTLS9HUYUcwNpwL2eroLtsZZ/w=="],
|
||||
"@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.219.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.219.0", "@opentelemetry/core": "2.8.0", "@opentelemetry/resources": "2.8.0", "@opentelemetry/sdk-logs": "0.219.0", "@opentelemetry/sdk-metrics": "2.8.0", "@opentelemetry/sdk-trace-base": "2.8.0" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-aaYKAyXhw9VchKZVGOopD3Gw/kPsyrX2c6IQ0AW32mTjqmZOh5Y6Gf5OYqTNqVktAeBjmFinhyFaCwW6GYK9YQ=="],
|
||||
|
||||
"@opentelemetry/resources": ["@opentelemetry/resources@2.6.1", "", { "dependencies": { "@opentelemetry/core": "2.6.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-lID/vxSuKWXM55XhAKNoYXu9Cutoq5hFdkbTdI/zDKQktXzcWBVhNsOkiZFTMU9UtEWuGRNe0HUgmsFldIdxVA=="],
|
||||
"@opentelemetry/resources": ["@opentelemetry/resources@2.8.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-qmXQ27ilDbUK/vGMqwL8D4/rhn76C+sherM4wTbjlfknR8Nvfc/hCxjRJPhkzZzUsPiNg16SA31NxMabwttRjg=="],
|
||||
|
||||
"@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.214.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.214.0", "@opentelemetry/core": "2.6.1", "@opentelemetry/resources": "2.6.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-zf6acnScjhsaBUU22zXZ/sLWim1dfhUAbGXdMmHmNG3LfBnQ3DKsOCITb2IZwoUsNNMTogqFKBnlIPPftUgGwA=="],
|
||||
"@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.219.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.219.0", "@opentelemetry/core": "2.8.0", "@opentelemetry/resources": "2.8.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-s6lTKRakaPClvKoWHRChxnXjDMkM/TQ30ff78jN6EBGf7MI7VzANE5PU3f4z9qDUudWjvZjOLHG0rBnBKYvoXA=="],
|
||||
|
||||
"@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.6.1", "", { "dependencies": { "@opentelemetry/core": "2.6.1", "@opentelemetry/resources": "2.6.1" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-9t9hJHX15meBy2NmTJxL+NJfXmnausR2xUDvE19XQce0Qi/GBtDGamU8nS1RMbdgDmhgpm3VaOu2+fiS/SfTpQ=="],
|
||||
"@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.8.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/resources": "2.8.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-UDBGaj6W0Rgy5rTTaoxs8gVGF/aGkAKyjurJv7se6wjRxJu7FoquTLT/vt54DZfo4crbprYfhX/SOK9+BPw1qg=="],
|
||||
|
||||
"@opentelemetry/sdk-trace": ["@opentelemetry/sdk-trace@2.11.0", "", { "dependencies": { "@opentelemetry/core": "2.11.0", "@opentelemetry/resources": "2.11.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-fFnTqGm8/G73GQVnxYi7LXa1ZVYEUvgL6XI1LpvV0bPC7WQ/ZGgKxCSl8FnlZBKto9JHHEFTO6s6CUpvvtwFrA=="],
|
||||
|
||||
"@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.6.1", "", { "dependencies": { "@opentelemetry/core": "2.6.1", "@opentelemetry/resources": "2.6.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-r86ut4T1e8vNwB35CqCcKd45yzqH6/6Wzvpk2/cZB8PsPLlZFTvrh8yfOS3CYZYcUmAx4hHTZJ8AO8Dj8nrdhw=="],
|
||||
"@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.8.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/resources": "2.8.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-mhU4jp+vW0mGbFRd+GeXHvmfA4aDqWjBjLC3pE5XMpLs0IE2ryYb019Ts2AQrOq67gaTF25D91+fgvEHDZEnuQ=="],
|
||||
|
||||
"@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.6.1", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.6.1", "@opentelemetry/core": "2.6.1", "@opentelemetry/sdk-trace-base": "2.6.1" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-Hh2i4FwHWRFhnO2Q/p6svMxy8MPsNCG0uuzUY3glqm0rwM0nQvbTO1dXSp9OqQoTKXcQzaz9q1f65fsurmOhNw=="],
|
||||
"@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.8.0", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.8.0", "@opentelemetry/core": "2.8.0", "@opentelemetry/sdk-trace-base": "2.8.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-nZt9OGufioAc3AfoLTqA9bsAeaMJAictYDdI2VcNQ+PmT+3rfKjAZDZvgPfd8VPX0O5Bw1hdQF6kDK8VSpZiWg=="],
|
||||
|
||||
"@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.43.0", "", {}, "sha512-eSYWTm620tTk45EKSedaUL8MFYI8hW164hIXsgIHyxu3VobUB3fFCu5t0hQby6OoWRPsG1KkKUG2M5UadiLiVg=="],
|
||||
|
||||
@@ -2553,24 +2558,6 @@
|
||||
|
||||
"@protobuf-ts/runtime-rpc": ["@protobuf-ts/runtime-rpc@2.11.1", "", { "dependencies": { "@protobuf-ts/runtime": "^2.11.1" } }, "sha512-4CqqUmNA+/uMz00+d3CYKgElXO9VrEbucjnBFEjqI4GuDrEQ32MaI3q+9qPBvIGOlL4PmHXrzM32vBPWRhQKWQ=="],
|
||||
|
||||
"@protobufjs/aspromise": ["@protobufjs/aspromise@1.1.2", "", {}, "sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ=="],
|
||||
|
||||
"@protobufjs/base64": ["@protobufjs/base64@1.1.2", "", {}, "sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg=="],
|
||||
|
||||
"@protobufjs/codegen": ["@protobufjs/codegen@2.0.5", "", {}, "sha512-zgXFLzW3Ap33e6d0Wlj4MGIm6Ce8O89n/apUaGNB/jx+hw+ruWEp7EwGUshdLKVRCxZW12fp9r40E1mQrf/34g=="],
|
||||
|
||||
"@protobufjs/eventemitter": ["@protobufjs/eventemitter@1.1.1", "", {}, "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg=="],
|
||||
|
||||
"@protobufjs/fetch": ["@protobufjs/fetch@1.1.1", "", { "dependencies": { "@protobufjs/aspromise": "^1.1.1" } }, "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw=="],
|
||||
|
||||
"@protobufjs/float": ["@protobufjs/float@1.0.2", "", {}, "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ=="],
|
||||
|
||||
"@protobufjs/path": ["@protobufjs/path@1.1.2", "", {}, "sha512-6JOcJ5Tm08dOHAbdR3GrvP+yUUfkjG5ePsHYczMFLq3ZmMkAD98cDgcT2iA1lJ9NVwFd4tH/iSSoe44YWkltEA=="],
|
||||
|
||||
"@protobufjs/pool": ["@protobufjs/pool@1.1.0", "", {}, "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw=="],
|
||||
|
||||
"@protobufjs/utf8": ["@protobufjs/utf8@1.1.2", "", {}, "sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug=="],
|
||||
|
||||
"@puppeteer/browsers": ["@puppeteer/browsers@3.2.1", "", { "dependencies": { "modern-tar": "^0.8.0", "yargs": "^18.0.0" }, "peerDependencies": { "proxy-agent": ">=8.0.1", "yauzl": "^2.10.0 || ^3.4.0" }, "optionalPeers": ["proxy-agent", "yauzl"], "bin": { "browsers": "lib/main-cli.js" } }, "sha512-KDz+3qDRdBAlRlMjmKyj6dEs33YHTk/xRHEENSXq6TNnhgoU15ruSHtEBeVF6OZ9tBDY55Se4P0nFMNsipzU9A=="],
|
||||
|
||||
"@radix-ui/colors": ["@radix-ui/colors@1.0.1", "", {}, "sha512-xySw8f0ZVsAEP+e7iLl3EvcBXX7gsIlC1Zso/sPBW9gIWerBTgz6axrjU+MZ39wD+WFi5h5zdWpsg3+hwt2Qsg=="],
|
||||
@@ -3881,13 +3868,13 @@
|
||||
|
||||
"ejs": ["ejs@3.1.10", "", { "dependencies": { "jake": "^10.8.5" }, "bin": { "ejs": "bin/cli.js" } }, "sha512-UeJmFfOrAQS8OJWPZ4qtgHyWExa088/MtK5UEyoJGFH67cDEXkZSviOiKRCZ4Xij0zxI3JECgYs3oKx+AizQBA=="],
|
||||
|
||||
"electron": ["electron@42.10.1", "", { "dependencies": { "@electron-internal/extract-zip": "^1.0.1", "@electron/get": "^5.0.0", "@types/node": "^24.9.0" }, "bin": { "electron": "cli.js", "install-electron": "install.js" } }, "sha512-ITc1HPeoDzsxCCaH6MFsN67Nq2nUiJf5N9pBWxzhUpFfKJF0IwiAkKT3emg1lPbvC4OfFE+Cdwe4vhgdOZ1YKg=="],
|
||||
"electron": ["electron@44.4.3", "", { "dependencies": { "@electron-internal/extract-zip": "^1.0.1", "@electron/get": "^5.0.0", "@types/node": "^24.9.0" }, "bin": { "electron": "cli.js", "install-electron": "install.js" } }, "sha512-LTpSFTB40qVCXIX5xMo+cgHI/Jjkbjw7VpB26PccEbroqOn72LBukeaDwPVo1fBYzSzs0c9iPuAucFCO7Tw81Q=="],
|
||||
|
||||
"electron-builder": ["electron-builder@26.15.7", "", { "dependencies": { "app-builder-lib": "26.15.7", "builder-util": "26.15.3", "builder-util-runtime": "9.7.0", "chalk": "^4.1.2", "ci-info": "^4.2.0", "dmg-builder": "26.15.7", "fs-extra": "^10.1.0", "lazy-val": "^1.0.5", "simple-update-notifier": "2.0.0", "yargs": "^17.6.2" }, "bin": { "electron-builder": "./cli.js", "install-app-deps": "./install-app-deps.js" } }, "sha512-DBpaNzxsPs1BvEblzFoNriSbzsBqDCy/gseIngeEhYzQG1IxfB7Hvc2tBBVmpWE2BTQGP9J1RrAvDT+Vc/uAxg=="],
|
||||
|
||||
"electron-builder-squirrel-windows": ["electron-builder-squirrel-windows@26.15.7", "", { "dependencies": { "app-builder-lib": "26.15.7", "builder-util": "26.15.3", "electron-winstaller": "5.4.0" } }, "sha512-B4uvn2NzFSuf084udWqugludFull6CRJiWe2dLzMnZLl6G5hdAGk0fsBMGlBSpKjvQCJn8IPc+S7OnJ+GXqwLA=="],
|
||||
|
||||
"electron-context-menu": ["electron-context-menu@4.1.2", "", { "dependencies": { "cli-truncate": "^4.0.0", "electron-dl": "^4.0.0", "electron-is-dev": "^3.0.1" } }, "sha512-9xYTUV0oRqKL50N9W71IrXNdVRB0LuBp3R1zkUdUc2wfIa2/QZwYYj5RLuO7Tn7ZSLVIaO3X6u+EIBK+cBvzrQ=="],
|
||||
"electron-context-menu": ["electron-context-menu@5.0.0", "", { "dependencies": { "cli-truncate": "^4.0.0", "electron-dl": "^4.0.0", "electron-is-dev": "^3.0.1" } }, "sha512-rgFpRtwY0/rhsRCoz9rE6VM4WueEsLbIpca7ucOhERVrbGB2dQrxa9xwBNLplU54jRgPuv6nTEbxoplE5bzy2A=="],
|
||||
|
||||
"electron-dl": ["electron-dl@4.0.0", "", { "dependencies": { "ext-name": "^5.0.0", "pupa": "^3.1.0", "unused-filename": "^4.0.1" } }, "sha512-USiB9816d2JzKv0LiSbreRfTg5lDk3lWh0vlx/gugCO92ZIJkHVH0UM18EHvKeadErP6Xn4yiTphWzYfbA2Ong=="],
|
||||
|
||||
@@ -4127,7 +4114,7 @@
|
||||
|
||||
"github-slugger": ["github-slugger@2.0.0", "", {}, "sha512-IaOQ9puYtjrkq7Y0Ygl9KDZnrf/aiUJYUpVf89y8kyaxbRG7Y1SrX/jaumrv81vc61+kiMempujsM3Yw7w5qcw=="],
|
||||
|
||||
"gitlab-ai-provider": ["gitlab-ai-provider@6.12.1", "", { "dependencies": { "@anthropic-ai/sdk": "^0.71.0", "@anycable/core": "^0.9.2", "graphql-request": "^6.1.0", "isomorphic-ws": "^5.0.0", "openai": "^6.16.0", "socket.io-client": "^4.8.1", "vscode-jsonrpc": "^8.2.1", "zod": "^3.25.76" }, "peerDependencies": { "@ai-sdk/provider": ">=3.0.0", "@ai-sdk/provider-utils": ">=4.0.0" } }, "sha512-Qn5iHqvjG8yktI5MWaUgdRR94l7O4WtYW0CAbhsCh1Tj0Fei/DeprOYPVyf4Nht1Ix6U2PXSYM32QOHI6Z2TDw=="],
|
||||
"gitlab-ai-provider": ["gitlab-ai-provider@6.16.0", "", { "dependencies": { "@anthropic-ai/sdk": "^0.71.0", "@anycable/core": "^0.9.2", "graphql-request": "^6.1.0", "isomorphic-ws": "^5.0.0", "openai": "^6.16.0", "socket.io-client": "^4.8.1", "vscode-jsonrpc": "^8.2.1", "zod": "^3.25.76" }, "peerDependencies": { "@ai-sdk/provider": ">=3.0.0", "@ai-sdk/provider-utils": ">=4.0.0" } }, "sha512-HMC3sKgWYaYSsgm86Cnq2e6laHlYkhiFQ6rFD5qsVghv9//6h4Ofr7j5R2KjQx9Hl8EBercPsmzjoEjEN7dX5Q=="],
|
||||
|
||||
"glob": ["glob@13.0.5", "", { "dependencies": { "minimatch": "^10.2.1", "minipass": "^7.1.2", "path-scurry": "^2.0.0" } }, "sha512-BzXxZg24Ibra1pbQ/zE7Kys4Ua1ks7Bn6pKLkVPZ9FZe4JQS6/Q7ef3LG1H+k7lUf5l4T3PLSyYyYJVYUvfgTw=="],
|
||||
|
||||
@@ -5047,8 +5034,6 @@
|
||||
|
||||
"proto-list": ["proto-list@1.2.4", "", {}, "sha512-vtK/94akxsTMhe0/cbfpR+syPuszcuwhqVjJq26CuNDgFGj682oRBXOP5MJpv2r7JtE8MsiepGIqvvOTBwn2vA=="],
|
||||
|
||||
"protobufjs": ["protobufjs@7.6.5", "", { "dependencies": { "@protobufjs/aspromise": "^1.1.2", "@protobufjs/base64": "^1.1.2", "@protobufjs/codegen": "^2.0.5", "@protobufjs/eventemitter": "^1.1.1", "@protobufjs/fetch": "^1.1.1", "@protobufjs/float": "^1.0.2", "@protobufjs/path": "^1.1.2", "@protobufjs/pool": "^1.1.0", "@protobufjs/utf8": "^1.1.1", "@types/node": ">=13.7.0", "long": "^5.3.2" } }, "sha512-/FPD0nUc9jH6rfFjji9IBqOz4pcSE3CsT1m7Ep6Mdb0LxSUMj8hgl6GomOvZzpNpAqqGaXA0P3VSrZLFzIhQrw=="],
|
||||
|
||||
"proxy-from-env": ["proxy-from-env@1.1.0", "", {}, "sha512-D+zkORCbA9f1tdWRK0RaCR3GPv50cMxcrz4X8k5LTSUD1Dkw47mKJEZQNunItRTkWwgtaUSo1RVFRIG9ZXiFYg=="],
|
||||
|
||||
"pump": ["pump@3.0.4", "", { "dependencies": { "end-of-stream": "^1.1.0", "once": "^1.3.1" } }, "sha512-VS7sjc6KR7e1ukRFhQSY5LM2uBWAUPiOPa/A3mkKmiMwSmRFUITt0xuj+/lesgnCv+dPIEYlkzrcyXgquIHMcA=="],
|
||||
@@ -6245,6 +6230,8 @@
|
||||
|
||||
"@opencode/www/wrangler": ["wrangler@4.110.0", "", { "dependencies": { "@cloudflare/kv-asset-handler": "0.5.0", "@cloudflare/unenv-preset": "2.16.1", "blake3-wasm": "2.1.5", "esbuild": "0.28.1", "miniflare": "4.20260708.1", "path-to-regexp": "6.3.0", "unenv": "2.0.0-rc.24", "workerd": "1.20260708.1" }, "optionalDependencies": { "fsevents": "2.3.3" }, "peerDependencies": { "@cloudflare/workers-types": "^5.20260708.1" }, "optionalPeers": ["@cloudflare/workers-types"], "bin": { "wrangler": "bin/wrangler.js", "wrangler2": "bin/wrangler.js", "cf-wrangler": "bin/cf-wrangler.js" } }, "sha512-xZeXKYi7hxQRF5anL+v77RkufJNpF9f3Eqeyqq2QBsETpLZgh0Agj0jJ6JPtkbgn6ukZdh8OK5egsGPWIditgg=="],
|
||||
|
||||
"@opentelemetry/api-logs/@opentelemetry/api": ["@opentelemetry/api@1.9.1", "", {}, "sha512-gLyJlPHPZYdAk1JENA9LeHejZe1Ti77/pTeFm/nMXmQH/HFZlcS/O2XJB+L8fkbrNSqhdtlvjBVjxwUYanNH5Q=="],
|
||||
|
||||
"@opentelemetry/instrumentation/@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.220.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CmVa4ImJ+ynfrPMNaAXHET6Bhb44SwzmfyVJFq9ni2jgXJR/l7C6gfVFddNmHP+ZOkP9cf4f9DBe68qVLTHc9w=="],
|
||||
|
||||
"@opentelemetry/sdk-trace/@opentelemetry/core": ["@opentelemetry/core@2.11.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-7YP44XH0tV6+Mb54x2YGf84i7yi+31MBZlE8JwvozkxyTvXbSp10X7cI7YE49ChJ3shMJoBmCJF3+1QFBJctGA=="],
|
||||
|
||||
+4
-4
@@ -1,8 +1,8 @@
|
||||
{
|
||||
"nodeModules": {
|
||||
"x86_64-linux": "sha256-TRfKunG6/UE8rQFyRkx/pX+pe7GT542IJWLEKcFFfAY=",
|
||||
"aarch64-linux": "sha256-JxCYBQoHeJVuEMEe4Ef5f6UxxUCAuhPo96jqMT047Fc=",
|
||||
"aarch64-darwin": "sha256-UEgjeQMivNC7JVsLEDlNbuqp9LJH84f9nHgLHTZ2zDI=",
|
||||
"x86_64-darwin": "sha256-jEs/Oadjf2LVAinY0xyFna0V+NTwoCKXiXmQ83OchF8="
|
||||
"x86_64-linux": "sha256-LQ1GAz1qF4R5P4j/kkgUygsQvxm7KStVlMf24nmyq44=",
|
||||
"aarch64-linux": "sha256-PsNR3VaClA1O1vSE0z/Hb3XRT1pb6PvNRXuK+giIx6s=",
|
||||
"aarch64-darwin": "sha256-7VCS+GT7tDYAMgqNCnLh1zvHAZkLTnJJ4vtfM88ruT8=",
|
||||
"x86_64-darwin": "sha256-HyxXjiFVd8vcKvSb/BuHAY5Jj3MGgdXEvuhDuglp4ss="
|
||||
}
|
||||
}
|
||||
|
||||
+2
-2
@@ -2,7 +2,7 @@
|
||||
"$schema": "https://json.schemastore.org/package.json",
|
||||
"name": "opencode",
|
||||
"description": "AI-powered development tool",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"packageManager": "bun@1.4.2",
|
||||
@@ -173,7 +173,7 @@
|
||||
"solid-js@1.9.15": "patches/solid-js@1.9.15.patch",
|
||||
"@ai-sdk/mistral@3.0.51": "patches/@ai-sdk%2Fmistral@3.0.51.patch",
|
||||
"gcp-metadata@8.1.2": "patches/gcp-metadata@8.1.2.patch",
|
||||
"pacote@21.5.0": "patches/pacote@21.5.0.patch",
|
||||
"pacote@21.5.1": "patches/pacote@21.5.1.patch",
|
||||
"@ai-sdk/google@3.0.73": "patches/@ai-sdk%2Fgoogle@3.0.73.patch",
|
||||
"@pierre/trees@1.0.0-beta.4": "patches/@pierre%2Ftrees@1.0.0-beta.4.patch",
|
||||
"@tanstack/virtual-core@3.17.8": "patches/@tanstack%2Fvirtual-core@3.17.8.patch",
|
||||
|
||||
+21
-2
@@ -10,7 +10,15 @@
|
||||
|
||||
## Conventions
|
||||
|
||||
Per-type constructors live on the type, not as top-level re-exports. Use `Message.system(...)`, `Message.user(...)`, `Message.assistant(...)`, `Message.tool(...)`, `LanguageModel.make(...)`, `ToolDefinition.make(...)`, `ToolCallPart.make(...)`, `ToolResultPart.make(...)`, `ToolChoice.make(...)`, `ToolChoice.named(...)`, `SystemPart.make(...)`, and `GenerationOptions.make(...)` directly. The top-level `LLM` namespace is reserved for request-shaped call APIs: `LLM.request`, `LLM.generate`, `LLM.stream`, and `LLM.generateObject`. Use `LLMRequest.update(...)` when deriving canonical request data; do not add a duplicate `LLM.updateRequest(...)` path. Two ways to construct the same thing is one too many.
|
||||
Per-type constructors live on the type, not as top-level re-exports. Use `Message.system(...)`, `Message.user(...)`, `Message.assistant(...)`, `Message.tool(...)`, `Message.media(...)`, `LanguageModel.make(...)`, `ToolDefinition.make(...)`, `ToolCallPart.make(...)`, `ToolResultPart.make(...)`, `ToolChoice.make(...)`, `ToolChoice.named(...)`, `SystemPart.make(...)`, and `GenerationOptions.make(...)` directly. The top-level `LLM` namespace is reserved for request-shaped call APIs: `LLM.request`, `LLM.generate`, `LLM.stream`, and `LLM.generateObject`. Use `LLMRequest.update(...)` when deriving canonical request data; do not add a duplicate `LLM.updateRequest(...)` path. Two ways to construct the same thing is one too many.
|
||||
|
||||
Modality namespaces mirror `LLM` exactly: `Image.request`, `Image.generate`, `Image.stream` (later `Video`, `Speech`, `Transcription`). Common request fields (`images`, `mask`, `n`, `size`, `aspectRatio`, `seed`, `format`) lower natively or fail with a typed `AIError`; provider-native controls always live under `providerOptions`, never under a modality-specific `options` key.
|
||||
|
||||
Media payloads are always `Media.Asset` (`src/media.ts`). Construct them with `Media.bytes`, `Media.base64`, `Media.url`, `Media.ref`, `Media.fromDataUrl`, or `Media.file`; never introduce a parallel `data: string | Uint8Array` shape. `MediaPart.media`, `ImageRequest.images`/`mask`, `ImageResponse.images`, and the `media` `LLMEvent` all share it. Protocols branch on `asset.source.type` and `asset.kind` and use `ProviderShared.inlineMedia` / `requireInlineMedia` / `mediaUrl` / `MediaInput.refID` rather than re-deriving base64 or URL handling.
|
||||
|
||||
`schema/messages.ts → media.ts → route/executor-service.ts` is an accepted runtime dependency from the schema layer on the executor service tag: `Media.Asset.bytes()` must be able to download `url` sources, and the tag lives in that leaf module precisely so the schema barrel never imports the executor implementation (which imports the schema barrel back). Do not move the tag into `route/executor.ts` or import `route/executor.ts` from `src/schema/*` or `src/media.ts`.
|
||||
|
||||
Nothing in `src/*` except `src/promise.ts` may know about Promises. `@opencode/ai/promise` (`AI.make({ layer? })`, default `ai`) is the single Promise/`AsyncIterable` surface for LLM and media; it runs the Effect APIs in one `ManagedRuntime` and rethrows `AIError` unchanged.
|
||||
|
||||
- Prefer forward compatibility for provider-defined options that OpenCode only passes through. For pass-through string enums, expose known values for autocomplete while accepting future values with `Known | (string & {})`, and accept any string at runtime. Closed literals are appropriate when OpenCode branches on a value, transforms its associated structure, or otherwise cannot correctly handle an unknown variant. New options whose shape or behavior requires implementation remain unsupported until they are handled; do not blindly forward unknown structures.
|
||||
- Order reasoning-effort values from lowest to highest: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Provider-specific subsets follow the same relative order in types, schemas, option lists, and tests.
|
||||
@@ -86,6 +94,16 @@ The four-axis decomposition is the reason DeepSeek, TogetherAI, Cerebras, Basete
|
||||
|
||||
When a provider supports multiple physical transports, selection remains execution policy below its semantic route. `OpenResponsesChannel.transport(...)` owns the provider-neutral Responses WebSocket concept: it prepares one final request, executes HTTP by default, strips WebSocket-disallowed fields, and passes a generic channel exchange to a per-call `WebSocketChannelExecutor` when supplied. Provider-specific Responses routes opt in with handshake and connection-age policy. `Route.streamPrepared` owns decoding and acknowledges channel completion only after successful full consumption.
|
||||
|
||||
### Media Routes
|
||||
|
||||
Media does not fit the SSE-frames-to-event-state-machine LLM route. `MediaRoute.make(...)` (`src/route/media.ts`) composes a `MediaProtocol` kind with `Endpoint` and `Auth` and owns the transport plumbing: `http` option merging, URL/query rendering, auth headers, JSON vs multipart encoding, and handing the response back to the protocol. `MediaProtocol.inline` (`src/route/media-protocol.ts`) is `body.from(request)` plus `response.decode(response, context)`; use `MediaProtocol.decodeJson` / `text` / `bytes` so decode failures retain the raw body and HTTP context. `Generation` (`src/generation.ts`) is the provider-neutral handle for a queued generation over a `GenerationRoute` (`status`, `result`, `cancel`, `pollHint`). Image protocol files follow the same section order as LLM protocols and declare unsupported common fields once through `MediaInput.rejectUnsupported`.
|
||||
|
||||
`MediaProtocol.queued` is the submit-then-poll kind every video route uses: `start` (body + decode into `{ token, snapshot }`), `status`, `result`, and optional `cancel`, each addressed by a route-owned `token` whose `Schema.Codec` makes it serializable. `MediaRoute.inline` and `MediaRoute.queued` compose the two kinds with `Endpoint` and `Auth`; the queued route decodes the token once at the boundary (`start` output or `resume` input) and closes over it in a token-free `GenerationRoute` (`status`/`result`/`cancel` are plain Effects), so `Generation` never sees the token's shape and only carries the encoded JSON for persistence. Polls reuse the route's auth and deployment headers plus the request's `http` overlay after `start`, and resolve relative paths against the route base URL (provider-issued absolute URLs such as fal's `status_url` pass through). `result` is always its own GET even when the provider returns output inside the status document, so `Generation.await` behaves the same after `start` and after `resume`. `PollContext.auth` carries only what `Auth` added so protocols can hand download credentials to output assets as transient `Media.Asset.headers` (Veo) — never part of `source` or JSON. Status strings map through a per-protocol `STATUS` table via `MediaProtocol.status`; terminal generations without output fail through `output.ended` / `output.contentPolicy` with the provider document on `reason.body`. `Generation.AwaitOptions` (`{ poll?: Poll }`) is the one options type for `await`, `events`, `Video.generate`, and `Video.stream`.
|
||||
|
||||
`MediaProtocol.stream` is the incremental kind every speech route uses, with the same discipline as LLM protocols. `MediaRoute.stream` submits the caller's request as `MediaProtocol.Addressed<Request>` (`{ ...request, mode }`, `mode: "generate" | "stream"`), so one provider stays one protocol: `body.from`, the endpoint path, and `frames` read `request.mode` to pick the body, path, and framing. `frames(bytes, context)` returns frames — `Framing.sse`, `Framing.lines`, `Framing.document` (a single-document response shaped like a streamed record), or the raw `bytes` for chunked audio. `initial()` is fresh per-response parser state; `step` folds each frame into it and emits modality events; `finish(state, context)` runs once after the last frame with the request, body, and observed `http` (header-only usage lives there) and emits exactly one terminal event or fails with `MediaProtocol.incomplete`. Keep parser state to real accumulators and derive anything the request or body determines in `finish`. `generate` runs the same stream and folds it with the modality's `collect`. Request-derived URL parameters go on the body's `query` (array values repeat the parameter), applied before route and caller `http.query`. Decode frames with `MediaProtocol.decodeFrame` and raise stream-time failures with `MediaProtocol.frameError` (the frame stays on `reason.body`); protocols never thread HTTP context, because the route fills `reason.http` on stream errors that lack it. Speech protocols share `protocols/utils/speech-stream.ts` for deltas, timestamps, voice ids, PCM and container descriptions, and the terminal asset.
|
||||
|
||||
Transcription uses all three kinds (OpenAI and Gemini stream, Deepgram is inline, AssemblyAI is queued): every route carries its `kind` and `TranscriptionClient` dispatches on it. Bodies are `json`, `multipart`, or `binary` (a raw upload), and a queued protocol that must upload media before submitting implements `start.prepare` (`MediaProtocol.Prepare`; AssemblyAI `/v2/upload`).
|
||||
|
||||
### URL Construction
|
||||
|
||||
`Endpoint` owns `{ baseURL, path, query }`. Each protocol route includes a canonical endpoint when the provider has one (e.g. `https://api.openai.com/v1`); provider helpers override endpoint fields by configuring the route before selecting a model. Generic OpenAI-compatible routes have no canonical URL and require configuration before execution.
|
||||
@@ -94,11 +112,12 @@ For providers where the URL is derived from typed inputs (Azure resource name, B
|
||||
|
||||
### Provider Facades
|
||||
|
||||
Provider-facing APIs are configured facades over route values. Endpoint/auth/resource/API-version setup happens before model selection, and model selectors accept only a model or deployment id:
|
||||
Provider-facing APIs are configured facades over route values. Endpoint/auth/resource/API-version setup happens before model selection, and model selectors accept only a model or deployment id. Media models use per-modality selectors on the same facade (`openai.image(id)`, later `.video` / `.speech` / `.transcription`) that mirror `openai.responses(id)`; the one-word overlap with the request namespace is accepted over a second construction path:
|
||||
|
||||
```ts
|
||||
const openai = OpenAI.configure({ apiKey, baseURL })
|
||||
const model = openai.responses("gpt-4o-mini")
|
||||
const image = openai.image("gpt-image-2")
|
||||
|
||||
const azure = Azure.configure({ resourceName, apiKey, apiVersion: "v1" })
|
||||
const deployment = azure.responses("my-deployment")
|
||||
|
||||
+383
-54
@@ -8,10 +8,10 @@ import { LLM, LLMClient } from "@opencode/ai"
|
||||
import { RequestExecutor } from "@opencode/ai/route"
|
||||
import { OpenAI } from "@opencode/ai/providers"
|
||||
|
||||
const model = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY }).responses("gpt-4o-mini")
|
||||
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
||||
|
||||
const request = LLM.request({
|
||||
model,
|
||||
model: openai.responses("gpt-4o-mini"), // `.chat(...)` selects the Chat Completions API instead
|
||||
system: "You are concise.",
|
||||
prompt: "Say hello in one short sentence.",
|
||||
generation: { maxTokens: 40 },
|
||||
@@ -29,6 +29,94 @@ await Effect.runPromise(program.pipe(Effect.provide(llmLayer)))
|
||||
|
||||
Run `LLMClient.stream(request)` instead of `generate` when you want incremental `LLMEvent`s. The event stream is provider-neutral — same shape across OpenAI Chat, OpenAI Responses, Anthropic Messages, Gemini, Bedrock Converse, and any OpenAI-compatible deployment.
|
||||
|
||||
The same configured facade names image models. `Image.request` resolves the provider's image route from the ref and
|
||||
returns `Media.Asset`s with lazily decoded bytes:
|
||||
|
||||
```ts
|
||||
import { NodeFileSystem } from "@effect/platform-node"
|
||||
import { Image, ImageClient, Media } from "@opencode/ai"
|
||||
|
||||
const image = Effect.gen(function* () {
|
||||
const response = yield* Image.generate({
|
||||
model: openai.image("gpt-image-2"),
|
||||
prompt: "A robot tending a rooftop garden",
|
||||
size: "1024x1024",
|
||||
providerOptions: { quality: "high" }, // typed per image model
|
||||
})
|
||||
yield* Media.write(response.image, "./garden.png")
|
||||
})
|
||||
|
||||
// `asset.bytes()` / `Media.write` also need the executor, so merge it into the environment instead of hiding it.
|
||||
const imageLayer = ImageClient.layer.pipe(Layer.provideMerge(RequestExecutor.fetchLayer))
|
||||
|
||||
await Effect.runPromise(image.pipe(Effect.provide(imageLayer), Effect.provide(NodeFileSystem.layer)))
|
||||
```
|
||||
|
||||
Prefer promises? `@opencode/ai/promise` exposes the same LLM and image APIs over one managed runtime:
|
||||
|
||||
```ts
|
||||
import { AI } from "@opencode/ai/promise"
|
||||
|
||||
const ai = AI.make()
|
||||
const text = await ai.llm.generate({ model: openai.responses("gpt-4o-mini"), prompt: "Say hello." })
|
||||
const generated = await ai.image.generate({ model: openai.image("gpt-image-2"), prompt: "A lighthouse" })
|
||||
for await (const event of ai.llm.stream({ model: openai.responses("gpt-4o-mini"), prompt: "Stream hello." })) {
|
||||
// LLMEvent
|
||||
}
|
||||
await ai.dispose()
|
||||
```
|
||||
|
||||
## Experimental evaluation
|
||||
|
||||
Evaluation models compare shared state with typed choice, score, and boolean questions. The API is
|
||||
isolated under an experimental entrypoint and provider namespace while the contract evolves:
|
||||
|
||||
```ts
|
||||
import { Effect } from "effect"
|
||||
import { Evaluation, EvaluationClient } from "@opencode/ai/experimental"
|
||||
import { TypeSafeAI } from "@opencode/ai/providers"
|
||||
|
||||
const model = TypeSafeAI.configure().experimental.evaluation("jev-latest")
|
||||
|
||||
const program = Evaluation.run({
|
||||
model,
|
||||
state: "I was charged twice. Please refund the duplicate payment.",
|
||||
questions: {
|
||||
department: {
|
||||
type: "choice",
|
||||
instructions: "Which team should handle this?",
|
||||
criteria: { billing: "Payments and refunds", technical: "Bugs and outages" },
|
||||
},
|
||||
urgency: {
|
||||
type: "score",
|
||||
instructions: "How urgent is this?",
|
||||
criteria: ["Can wait", "Needs prompt attention", "Blocking revenue"],
|
||||
},
|
||||
refund: { type: "boolean", instructions: "Is the customer asking for a refund?" },
|
||||
},
|
||||
})
|
||||
|
||||
const response = await Effect.runPromise(program.pipe(Effect.provide(EvaluationClient.fetchLayer)))
|
||||
|
||||
console.log(response.answers.department.choice)
|
||||
console.log(response.answers.refund.probability)
|
||||
```
|
||||
|
||||
`TypeSafeAI` reads `TYPESAFE_API_KEY`. `OpenCodeZen` exposes the same selector and reads
|
||||
`OPENCODE_API_KEY`. OpenRouter and Vercel AI Gateway use the same provider shape:
|
||||
|
||||
```ts
|
||||
import { OpenRouter, VercelAIGateway } from "@opencode/ai/providers"
|
||||
|
||||
OpenRouter.configure().experimental.evaluation("typesafe/jev-1.13")
|
||||
VercelAIGateway.configure().experimental.evaluation("typesafe-ai/jev")
|
||||
```
|
||||
|
||||
OpenRouter reads `OPENROUTER_API_KEY`. Vercel reads `AI_GATEWAY_API_KEY`, then `VERCEL_OIDC_TOKEN`.
|
||||
The common API uses `boolean`; System One routes lower it to native `noul`.
|
||||
Choice and score confidence plus score legends remain available in provider metadata, and the
|
||||
provider's rounded probabilities are returned unchanged.
|
||||
|
||||
## Alibaba Cloud Model Studio
|
||||
|
||||
`Alibaba` provides standard Model Studio inference. Configure a region explicitly, then select
|
||||
@@ -314,23 +402,25 @@ citations or separate result blocks. Retain `response.message` for either API's
|
||||
Use `Image.generate` for one-off generation or editing:
|
||||
|
||||
```ts
|
||||
import { Image, ImageInput } from "@opencode/ai"
|
||||
import { Image, Media } from "@opencode/ai"
|
||||
|
||||
const generation = Image.generate({
|
||||
model: meta.image("muse-image-1.0"),
|
||||
model: meta("muse-image-1.0"),
|
||||
prompt: "A flat black square on a white background.",
|
||||
options: { n: 1, reasoningStrength: "low" },
|
||||
n: 1,
|
||||
providerOptions: { reasoningStrength: "low" },
|
||||
})
|
||||
|
||||
const edit = Image.generate({
|
||||
model: meta.image("muse-image-1.0"),
|
||||
model: meta("muse-image-1.0"),
|
||||
prompt: "Make the square purple.",
|
||||
images: [ImageInput.bytes(imageBytes, "image/webp")],
|
||||
options: { outputFormat: "png", reasoningStrength: "low" },
|
||||
images: [Media.bytes(imageBytes, "image/webp")],
|
||||
format: "png",
|
||||
providerOptions: { reasoningStrength: "low" },
|
||||
})
|
||||
```
|
||||
|
||||
The default image format is WEBP; `outputFormat` also accepts PNG/JPEG and `responseFormat: "url"`
|
||||
The default image format is WEBP; `format` also accepts PNG/JPEG and `responseFormat: "url"`
|
||||
returns a signed URL. `size` is an aspect-ratio hint. For conversational images, select
|
||||
`meta.responses("muse-image-1.0")` with `tools: [Meta.imageGeneration({ reasoningStrength: "low" })]`.
|
||||
Generated images are provider-executed tool results with file content. Retain `response.message` to
|
||||
@@ -341,29 +431,40 @@ Meta Responses is explicitly HTTP/SSE-only and does not use WebSockets, even whe
|
||||
|
||||
## Image generation
|
||||
|
||||
Use `Image.generate` with an image model for direct asset generation:
|
||||
Use `Image.generate` with an image model for direct asset generation. `Image.request` mirrors `LLM.request`: the
|
||||
model comes from the facade's `.image(...)` selector (mirroring `.responses(...)`), common fields
|
||||
(`images`, `mask`, `n`, `size`, `aspectRatio`, `seed`, `format`) lower natively or fail typed, and
|
||||
`providerOptions` is inferred from the selected model:
|
||||
|
||||
```ts
|
||||
import { Image, ImageInput } from "@opencode/ai"
|
||||
import { Image, Media } from "@opencode/ai"
|
||||
import { OpenAI } from "@opencode/ai/providers"
|
||||
|
||||
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
||||
|
||||
const program = Effect.gen(function* () {
|
||||
const response = yield* Image.generate({
|
||||
model: OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY }).image("gpt-image-2"),
|
||||
model: openai.image("gpt-image-2"),
|
||||
prompt: "A robot tending a rooftop garden",
|
||||
options: {
|
||||
n: 2,
|
||||
size: "1024x1024",
|
||||
n: 2,
|
||||
size: "1024x1024",
|
||||
format: "webp",
|
||||
providerOptions: {
|
||||
quality: "high", // inferred from the OpenAI image model
|
||||
outputFormat: "webp",
|
||||
future_option: true, // unknown native options pass through unchanged
|
||||
},
|
||||
})
|
||||
|
||||
return response.images // GeneratedImage[] with owned bytes or a provider URL
|
||||
return response.images // Media.Asset[] with owned bytes or a provider URL
|
||||
})
|
||||
```
|
||||
|
||||
`Media.Asset` is the one asset type shared by image requests, image responses, LLM messages, and tool results.
|
||||
`asset.source` is the serializable `Media.Source` (`bytes`, `base64`, `url`, or `ref`); `asset.bytes()`,
|
||||
`asset.base64()`, and `asset.dataUrl()` decode or download lazily and cache; `asset.materialize()` pulls a `url`
|
||||
asset into owned bytes before the provider URL expires. Construct assets with `Media.bytes`, `Media.base64`,
|
||||
`Media.url`, `Media.ref(provider, id)`, `Media.fromDataUrl`, or `Media.file(path)`.
|
||||
|
||||
Pass ordered image inputs to the same method for editing, composition, or image-conditioned generation:
|
||||
|
||||
```ts
|
||||
@@ -373,49 +474,45 @@ const response =
|
||||
model,
|
||||
prompt: "Combine these product photos into one studio scene",
|
||||
images: [
|
||||
ImageInput.bytes(firstBytes, "image/png"),
|
||||
ImageInput.url("https://example.com/second.webp"),
|
||||
ImageInput.file("file_123"),
|
||||
Media.bytes(firstBytes, "image/png"),
|
||||
Media.url("https://example.com/second.webp"),
|
||||
Media.ref("openai", "file_123"),
|
||||
],
|
||||
options,
|
||||
providerOptions,
|
||||
http,
|
||||
})
|
||||
```
|
||||
|
||||
`ImageInput.fileUri(uri, mediaType)` represents provider file URIs such as Gemini Files. Raw strings are not
|
||||
accepted as image inputs, avoiding ambiguity between base64, URLs, and provider IDs. Empty or omitted `images`
|
||||
uses text-to-image generation; a non-empty array selects the provider's edit behavior without enforcing provider
|
||||
image-count limits locally. `images` is the only common image-editing field. OpenAI uses multipart for byte/data-URL
|
||||
edits and its JSON reference body for URL or file-ID edits. Its provider-specific `options.mask` accepts an
|
||||
`ImageInput` for inpainting:
|
||||
`Media.ref(provider, id)` represents provider file handles such as OpenAI file IDs or Gemini Files URIs; routes
|
||||
only forward refs that belong to their own provider. Raw strings are not accepted as image inputs, avoiding
|
||||
ambiguity between base64, URLs, and provider IDs. Empty or omitted `images` uses text-to-image generation; a
|
||||
non-empty array selects the provider's edit behavior without enforcing provider image-count limits locally. OpenAI
|
||||
uses multipart for byte/data-URL edits and its JSON reference body for URL or file-ID edits. The common `mask`
|
||||
field selects inpainting; routes that cannot honor it fail with `UnsupportedOperation`:
|
||||
|
||||
```ts
|
||||
yield *
|
||||
Image.generate({
|
||||
model: OpenAI.configure({ apiKey }).image("gpt-image-2"),
|
||||
model: openai.image("gpt-image-2"),
|
||||
prompt,
|
||||
images: [ImageInput.bytes(sourceBytes, "image/png")],
|
||||
options: { mask: ImageInput.bytes(maskBytes, "image/png") },
|
||||
images: [Media.bytes(sourceBytes, "image/png")],
|
||||
mask: Media.bytes(maskBytes, "image/png"),
|
||||
})
|
||||
```
|
||||
|
||||
The OpenAI adapter extracts this helper value into the edit request's native `mask` field rather than passing the
|
||||
tagged `ImageInput` object through as an ordinary option. On multipart requests, `http.body` can override option
|
||||
fields but not structural `model`, `prompt`, `image[]`, or `mask` fields, and the transport owns the multipart
|
||||
`Content-Type` boundary. For JSON requests, `http.body` remains the final raw-native overlay. Gemini does not fetch
|
||||
public HTTP URLs, and hosted Z.ai image generation does not accept image inputs. These cases fail with
|
||||
`InvalidRequest` before network I/O.
|
||||
On multipart requests, `http.body` can override option fields but not structural `model`, `prompt`, `image[]`,
|
||||
or `mask` fields, and the transport owns the multipart `Content-Type` boundary. For JSON requests, `http.body`
|
||||
remains the final raw-native overlay. Gemini does not fetch public HTTP URLs, and hosted Z.ai image generation does
|
||||
not accept image inputs. These cases fail with a typed `AIError` before network I/O.
|
||||
|
||||
Provider-native image options belong to each request. Raw `http.body` fields have final precedence over them:
|
||||
|
||||
```ts
|
||||
const model = OpenAI.configure({ apiKey }).image("gpt-image-2")
|
||||
|
||||
yield *
|
||||
Image.generate({
|
||||
model,
|
||||
model: openai.image("gpt-image-2"),
|
||||
prompt,
|
||||
options: { quality: "medium" },
|
||||
providerOptions: { quality: "medium" },
|
||||
http,
|
||||
})
|
||||
```
|
||||
@@ -425,11 +522,11 @@ xAI image models use the same request API with xAI-native controls:
|
||||
```ts
|
||||
yield *
|
||||
Image.generate({
|
||||
model: XAI.configure({ apiKey }).image("any-model-id"),
|
||||
model: XAI.configure({ apiKey })("any-model-id"),
|
||||
prompt,
|
||||
options: {
|
||||
n: 2,
|
||||
aspectRatio: "16:9",
|
||||
n: 2,
|
||||
aspectRatio: "16:9",
|
||||
providerOptions: {
|
||||
resolution: "1k",
|
||||
responseFormat: "b64_json",
|
||||
future_option: true,
|
||||
@@ -445,12 +542,12 @@ import { Google } from "@opencode/ai/providers"
|
||||
|
||||
const googleProgram = Effect.gen(function* () {
|
||||
const response = yield* Image.generate({
|
||||
model: Google.configure({ apiKey }).image("any-model-id"),
|
||||
model: Google.configure({ apiKey })("any-model-id"),
|
||||
prompt: "A robot tending a rooftop garden",
|
||||
options: {
|
||||
aspectRatio: "16:9",
|
||||
aspectRatio: "16:9",
|
||||
seed: 42,
|
||||
providerOptions: {
|
||||
imageSize: "2K",
|
||||
seed: 42,
|
||||
thinkingLevel: "HIGH",
|
||||
includeThoughts: true,
|
||||
futureOption: true,
|
||||
@@ -472,9 +569,9 @@ Z.ai image models infer open Z.ai-native options from the selected model:
|
||||
```ts
|
||||
yield *
|
||||
Image.generate({
|
||||
model: ZAI.configure({ apiKey }).image("any-model-id"),
|
||||
model: ZAI.configure({ apiKey })("any-model-id"),
|
||||
prompt,
|
||||
options: {
|
||||
providerOptions: {
|
||||
quality: "hd",
|
||||
userID: "user-123",
|
||||
future_option: true,
|
||||
@@ -484,8 +581,8 @@ yield *
|
||||
```
|
||||
|
||||
Z.ai does not include trustworthy MIME metadata for output URLs, so generated images use
|
||||
`application/octet-stream`. Output URLs expire after 30 days; download and persist them promptly if they must
|
||||
remain available.
|
||||
`application/octet-stream` until materialized. Output URLs expire after 30 days; call `asset.materialize()` and
|
||||
persist the bytes promptly if they must remain available.
|
||||
|
||||
Conversational image generation remains part of the LLM interaction. OpenAI Responses exposes it through its hosted image tool:
|
||||
|
||||
@@ -503,7 +600,234 @@ const program = Effect.gen(function* () {
|
||||
})
|
||||
```
|
||||
|
||||
The hosted result is represented as a provider-executed tool call and tool result. Its image is a `file` content item with a data URI, so retaining `response.message` preserves the generated image for continuation.
|
||||
The hosted result is represented as a provider-executed tool call and tool result, and the generated image is also emitted as a first-class `media` `LLMEvent` (`response.message` then carries a `media` part). Gemini image-capable models emit the same `media` event for inline image output. Retaining `response.message` preserves the generated image for continuation on both routes.
|
||||
|
||||
## Video generation
|
||||
|
||||
Video mirrors `Image` with one difference: every provider is asynchronous, so the route is a submit-then-poll
|
||||
`Generation`. Models come from `.video(...)` selectors on the `Google` (Veo), `XAI`, `Fal`, and `Runway` facades.
|
||||
Common fields (`frames`, `references`, `video`, `durationSeconds`, `aspectRatio`, `resolution`, `audio`, `n`, `seed`,
|
||||
`negativePrompt`) lower natively or fail with a typed `AIError` before any network call; provider-native controls live
|
||||
under `providerOptions`, inferred from the selected model.
|
||||
|
||||
```ts
|
||||
import { Video, VideoClient } from "@opencode/ai"
|
||||
import { Google } from "@opencode/ai/providers"
|
||||
|
||||
const google = Google.configure({ apiKey: process.env.GOOGLE_GENERATIVE_AI_API_KEY })
|
||||
|
||||
// Simple: submit and wait.
|
||||
const program = Effect.gen(function* () {
|
||||
const response = yield* Video.generate(
|
||||
{
|
||||
model: google.video("veo-3.1-generate-preview"),
|
||||
prompt: "Panning wide shot of a calico kitten sleeping in the sunshine",
|
||||
aspectRatio: "16:9",
|
||||
resolution: "1080p",
|
||||
durationSeconds: 8,
|
||||
providerOptions: { personGeneration: "allow_adult" },
|
||||
},
|
||||
{ poll: { interval: "10 seconds", timeout: "10 minutes" } },
|
||||
)
|
||||
// Veo serves files for two days behind the API key. The asset knows the deadline (`expiresAt`) and carries the
|
||||
// download credentials only on the live instance (`asset.headers`), never in `source` or JSON: materialize
|
||||
// before persisting, or the persisted URL cannot be fetched again.
|
||||
return yield* response.video.materialize()
|
||||
})
|
||||
|
||||
// Explicit control: keep the handle, persist the token, resume elsewhere.
|
||||
const controlled = Effect.gen(function* () {
|
||||
const generation = yield* Video.start({ model: google.video("veo-3.1-generate-preview"), prompt })
|
||||
generation.id // provider operation / task / request id
|
||||
generation.status // "queued" | "running" | "completed" | "failed" | "cancelled" | "expired"
|
||||
generation.token // route-owned JSON: `{ operation }`, `{ requestID }`, `{ taskID }`, or fal's follow-up URLs
|
||||
const saved = JSON.stringify(generation.token)
|
||||
|
||||
const resumed = yield* Video.resume(google.video("veo-3.1-generate-preview"), JSON.parse(saved))
|
||||
return yield* resumed.await({ poll: { interval: "10 seconds" } })
|
||||
})
|
||||
|
||||
// Progress as a stream: generation-queued | generation-progress | video | finish.
|
||||
const events = Video.stream({ model: Runway.configure({ apiKey }).video("gen4.5"), prompt }, { poll })
|
||||
```
|
||||
|
||||
`VideoClient.layer` needs `RequestExecutor.Service`, and status polls, result fetches, cancels, and asset downloads
|
||||
all run through the same executor with the route's auth. `Generation.await` and `Generation.events` fail with a
|
||||
`Timeout` reason when `poll.timeout` (default 10 minutes) elapses. Failed,
|
||||
cancelled, and expired generations fail typed with the provider's terminal document on `reason.body`; moderation
|
||||
outcomes (Veo `raiMediaFilteredReasons`, xAI `respect_moderation`, Runway `SAFETY.*` codes) surface as `notices` when
|
||||
a video is still returned and as a `ContentPolicy` reason when nothing is.
|
||||
|
||||
Provider notes:
|
||||
|
||||
- **Google Veo** takes inline bytes only (materialize `url` assets first); `frames.last` requires `frames.first`;
|
||||
audio is always on, so `audio: false` fails typed; one video per request. Output URLs need the API key to
|
||||
download, which the returned asset holds transiently (see above).
|
||||
- **xAI** sends a `video` input to `/videos/edits`, or `/videos/extensions` with `providerOptions.mode: "extend"`.
|
||||
`seed` and `negativePrompt` are not supported.
|
||||
- **fal** endpoints are model-specific: `durationSeconds`, `references`, and `frames.last` fail typed and belong in
|
||||
`providerOptions` under the model's own names (`duration: "8s"`, `end_image_url`, …). Auth is
|
||||
`Authorization: Key <FAL_KEY>`.
|
||||
- **Runway** expects pixel ratios in `aspectRatio` for most models (`"1280:720"`), pins `X-Runway-Version`, reports
|
||||
`usage: { type: "credits" }`, and its output URLs expire after 24–48 hours.
|
||||
|
||||
The promise client exposes the same surface: `ai.video.start(...)` resolves to a handle with `await`, `refresh`,
|
||||
`cancel`, and `token`; `ai.video.generate`, `ai.video.resume(model, token)`, and `ai.video.stream` mirror the Effect
|
||||
API.
|
||||
|
||||
```ts
|
||||
import { ai } from "@opencode/ai/promise"
|
||||
|
||||
const generation = await ai.video.start({ model, prompt })
|
||||
const video = await generation.await({ poll: { interval: 10_000 }, signal })
|
||||
```
|
||||
|
||||
## Speech generation
|
||||
|
||||
Speech (text-to-speech) is one request whose response is parsed incrementally, so every route supports both
|
||||
`Speech.generate` (the whole file) and `Speech.stream` (audio chunks as they arrive). Models come from `.speech(...)`
|
||||
selectors on the `OpenAI`, `Google` (Gemini TTS), `ElevenLabs`, `Cartesia`, and `Deepgram` facades. Common fields
|
||||
(`voice`, `format`, `speed`, `language`, `instructions`, `timestamps`) lower natively or fail with a typed `AIError`
|
||||
before any network call; provider-native controls live under `providerOptions`, inferred from the selected model.
|
||||
|
||||
```ts
|
||||
import { Media, Speech, SpeechClient, SpeechEvent } from "@opencode/ai"
|
||||
import { ElevenLabs, OpenAI } from "@opencode/ai/providers"
|
||||
|
||||
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
||||
|
||||
// The whole file, written to disk.
|
||||
const program = Effect.gen(function* () {
|
||||
const response = yield* Speech.generate({
|
||||
model: openai.speech("gpt-4o-mini-tts"),
|
||||
text: "Hello from OpenCode.",
|
||||
voice: "coral",
|
||||
format: "mp3",
|
||||
instructions: "Warm and unhurried.",
|
||||
})
|
||||
response.audio // Media.Asset with bytes; headerless PCM carries info.encoding / sampleRate / channels
|
||||
response.usage // undefined: OpenAI reports tokens only on SSE streams (Gemini: tokens; ElevenLabs: credits; Deepgram: characters)
|
||||
yield* Media.write(response.audio, "hello.mp3")
|
||||
})
|
||||
|
||||
// Chunks as they arrive: audio-delta* (interleaved with timestamps) then one finish carrying the assembled asset.
|
||||
const events = Speech.stream({
|
||||
model: ElevenLabs.configure({ apiKey }).speech("eleven_flash_v2_5"),
|
||||
text: "Hello from OpenCode.",
|
||||
voice: "JBFqnCBsd6RMkjVDRZzb",
|
||||
format: "pcm",
|
||||
timestamps: true,
|
||||
}).pipe(
|
||||
Stream.tap((event) => {
|
||||
if (SpeechEvent.is.audioDelta(event)) return play(event.chunk)
|
||||
if (SpeechEvent.is.timestamps(event)) return highlight(event.items) // { text, startSeconds, endSeconds }[]
|
||||
return Effect.void
|
||||
}),
|
||||
)
|
||||
```
|
||||
|
||||
`voice` is the provider's own identifier — a name on OpenAI and Gemini (`"coral"`, `"Kore"`), a voice id on
|
||||
ElevenLabs and Cartesia. `{ id }` selects an OpenAI custom voice (`{ id: "voice_1234" }`) and means the same as the
|
||||
plain string elsewhere. There is no cross-provider voice catalog. `format` is the container-level word (`mp3`, `wav`,
|
||||
`pcm`, `opus`, `aac`, `flac`); sample rates and bitrates live under `providerOptions`, and a value the route cannot
|
||||
produce fails as `UnsupportedOperation`. Streams buffer every chunk so `finish` can carry the whole clip.
|
||||
`SpeechClient.layer` needs `RequestExecutor.Service`.
|
||||
|
||||
Provider notes:
|
||||
|
||||
- **OpenAI** streams over SSE (`stream_format: "sse"`), which is also the only place it reports token usage; `tts-1`
|
||||
and `tts-1-hd` do not support SSE and stream the raw audio body instead. `pcm` is 24 kHz 16-bit mono. `language`
|
||||
and `timestamps` are not supported.
|
||||
- **Gemini TTS** returns raw 16-bit PCM only (`audio/L16;codec=pcm;rate=24000`), so any `format` other than `pcm`
|
||||
fails typed; wrap the samples yourself. Style is directed in the text, so `instructions` and `speed` fail typed.
|
||||
Only `gemini-3.1-flash-tts-preview` and later support streaming. Two-speaker audio goes through
|
||||
`providerOptions.speechConfig.multiSpeakerVoiceConfig`.
|
||||
- **ElevenLabs** requires `voice` (the path voice id) and authenticates with `xi-api-key`. `format` maps to the
|
||||
`output_format` query parameter (`mp3_44100_128`, `pcm_24000`, `wav_24000`, `opus_48000_64`);
|
||||
`providerOptions.outputFormat` sets the exact string. WAV is only available from `generate`. `timestamps: true`
|
||||
selects the `with-timestamps` endpoints and yields character-level alignment. `instructions` is not supported.
|
||||
- **Cartesia** requires `voice` and pins `Cartesia-Version`. `generate` defaults to MP3 from `/tts/bytes`; streams
|
||||
and `timestamps: true` (word-level) use `/tts/sse`, which only serves raw PCM. `providerOptions.sampleRate`,
|
||||
`bitRate`, and `encoding` complete `output_format`. No usage is reported.
|
||||
- **Deepgram** Aura's voice is the model id (`aura-2-thalia-en`), so `voice` and `language` fail typed. `format`
|
||||
and `providerOptions` lower to query parameters (`encoding`, `container`, `sample_rate`, `bit_rate`); `pcm` is
|
||||
`linear16` without a container. Auth is `Authorization: Token <DEEPGRAM_API_KEY>`.
|
||||
|
||||
The promise client mirrors the Effect API; `ai.speech.stream` is an `AsyncIterable`.
|
||||
|
||||
```ts
|
||||
import { ai } from "@opencode/ai/promise"
|
||||
|
||||
const response = await ai.speech.generate({ model, text: "Hello from OpenCode.", voice: "coral" })
|
||||
await Bun.write("hello.mp3", await ai.run(response.audio.bytes()))
|
||||
|
||||
for await (const event of ai.speech.stream({ model, text: "Hello from OpenCode.", voice: "coral" })) {
|
||||
if (event.type === "audio-delta") player.write(event.chunk)
|
||||
}
|
||||
```
|
||||
|
||||
## Transcription
|
||||
|
||||
Transcription (speech-to-text) is the one modality whose providers use every route kind: OpenAI and Gemini stream,
|
||||
Deepgram answers inline, and AssemblyAI is queued. `Transcription.generate` and `Transcription.stream` work on all of
|
||||
them; `Transcription.start` / `resume` return a `Generation` on queued routes and fail with `UnsupportedOperation`
|
||||
elsewhere. Models come from `.transcription(...)` selectors on the `OpenAI`, `Google`, `Deepgram`, and `AssemblyAI`
|
||||
facades. Common fields (`language`, `prompt`, `timestamps: "none" | "segment" | "word"`, `diarize`, `speakers`) lower
|
||||
natively or fail with a typed `AIError` before any network call; a route may return more than asked.
|
||||
|
||||
```ts
|
||||
import { Media, Transcription, TranscriptionEvent } from "@opencode/ai"
|
||||
import { AssemblyAI, Deepgram, OpenAI } from "@opencode/ai/providers"
|
||||
|
||||
const openai = OpenAI.configure({ apiKey: process.env.OPENAI_API_KEY })
|
||||
|
||||
const program = Effect.gen(function* () {
|
||||
const audio = yield* Media.file("./call.mp3")
|
||||
|
||||
// Speaker-labelled segments; labels are provider-native strings ("A", "0", "spk:0").
|
||||
const response = yield* Transcription.generate({
|
||||
model: Deepgram.configure({ apiKey }).transcription("nova-3"),
|
||||
audio,
|
||||
diarize: true,
|
||||
timestamps: "word",
|
||||
})
|
||||
response.text // "Hello from OpenCode."
|
||||
response.segments // [{ text, startSeconds, endSeconds, speaker: "0" }]
|
||||
response.words // [{ text, startSeconds, endSeconds, speaker, confidence }]
|
||||
response.language // the provider's own value, lowercased ("en", "english", "en_us")
|
||||
|
||||
// Text deltas as the model transcribes, then one finish carrying the whole transcript.
|
||||
yield* Transcription.stream({ model: openai.transcription("gpt-4o-mini-transcribe"), audio }).pipe(
|
||||
Stream.tap((event) => (TranscriptionEvent.is.textDelta(event) ? Console.log(event.delta) : Effect.void)),
|
||||
Stream.runDrain,
|
||||
)
|
||||
|
||||
// Queued: persist the token, resume from another process, and await.
|
||||
const model = AssemblyAI.configure({ apiKey }).transcription("universal-3-5-pro")
|
||||
const generation = yield* Transcription.start({ model, audio })
|
||||
const resumed = yield* Transcription.resume(model, JSON.parse(JSON.stringify(generation.token)))
|
||||
const transcript = yield* resumed.await({ poll: { interval: "3 seconds" } })
|
||||
})
|
||||
```
|
||||
|
||||
Inline routes emit only `finish` from `stream` (no faked deltas); queued routes emit `generation-queued` /
|
||||
`generation-progress` before it. `TranscriptionClient.layer` needs `RequestExecutor.Service`.
|
||||
|
||||
Provider notes:
|
||||
|
||||
- **OpenAI** takes inline audio only; `diarize` needs `gpt-4o-transcribe-diarize`, timestamps need `whisper-1`, and `whisper-1` does not stream.
|
||||
- **Gemini** needs a transcribe model (`gemini-3.5-transcribe`); `prompt` and `speakers` fail typed.
|
||||
- **Deepgram** detects the language unless `language` is set; vocabulary goes in `providerOptions.keyterm`.
|
||||
- **AssemblyAI** uploads inline audio before submitting and is the only route that accepts `speakers`.
|
||||
|
||||
The promise client mirrors the Effect API:
|
||||
|
||||
```ts
|
||||
const text = (await ai.transcription.generate({ model, audio })).text
|
||||
for await (const event of ai.transcription.stream({ model, audio })) if (event.type === "text-delta") write(event.delta)
|
||||
const generation = await ai.transcription.start({ model: assemblyai, audio })
|
||||
const transcript = await generation.await({ poll: { interval: 3_000 } })
|
||||
```
|
||||
|
||||
## Public API
|
||||
|
||||
@@ -512,8 +836,13 @@ The hosted result is represented as a provider-executed tool call and tool resul
|
||||
- **`Message.user(...)` / `Message.assistant(...)` / `Message.tool(...)`** — message constructors from the canonical schema model.
|
||||
- **`LanguageModel.make(...)` / `ToolCallPart.make(...)` / `ToolResultPart.make(...)` / `ToolDefinition.make(...)`** — model and tool-related constructors from the canonical schema model.
|
||||
- **`LLMEvent.is.*`** — typed guards (`is.textDelta`, `is.toolCall`, `is.finish`, …) for filtering streams.
|
||||
- **`Image.generate({...})`** — generate images through a provider-neutral image request and response model.
|
||||
- **`Image.request` / `Image.generate` / `Image.stream`** — generate images through a provider-neutral image request and response model.
|
||||
- **`ImageClient`** — Effect service and layer for image execution, parallel to `LLMClient`.
|
||||
- **`Media`** — the shared asset type (`Media.Asset`, `Media.Source`) and constructors used by messages, tool results, and media requests.
|
||||
- **`Generation`** — provider-neutral handle for an in-flight media generation (`await`, `refresh`, `cancel`, `events`) used by queued media routes.
|
||||
- **`Speech.request` / `Speech.generate` / `Speech.stream`** — text-to-speech through a provider-neutral request; `SpeechClient` is its Effect service and layer.
|
||||
- **`Transcription.request` / `generate` / `stream` / `start` / `resume`** — speech-to-text over inline, streaming, and queued routes; `TranscriptionClient` is its Effect service and layer.
|
||||
- **`@opencode/ai/promise`** — `AI.make({ layer? })` and a default `ai` client exposing `llm`, `image`, `video`, `speech`, and `transcription` as Promise / `AsyncIterable` APIs.
|
||||
|
||||
## Testing
|
||||
|
||||
|
||||
@@ -0,0 +1,444 @@
|
||||
# Media generation in `@opencode/ai` — public API direction
|
||||
|
||||
Status: phases 1–3 implemented (Speech and Transcription); phases 4–5 proposal.
|
||||
|
||||
## Goal
|
||||
|
||||
`@opencode/ai` becomes the one package you reach for to generate anything: text, images, video, speech, transcripts, and later music and realtime. The LLM surface already exists and is shaped by three constraints: Effect-first, used by OpenCode Core, usable externally. Media has a different priority order: **external DX first**, Effect and Promise as peers, Core as one consumer among many.
|
||||
|
||||
The design below is derived from a survey of the raw provider APIs (OpenAI, Gemini/Veo/Imagen, xAI, Stability, BFL, fal, Replicate, Runway, Luma, Kling, MiniMax, ElevenLabs, Deepgram, Cartesia, AssemblyAI, Lyria) and of existing multi-provider SDKs.
|
||||
|
||||
## What the survey forces
|
||||
|
||||
1. **Three execution shapes, everywhere.** Inline sync (OpenAI images, all TTS, Gemini), async job with polling or webhook (every video provider, BFL, fal, Replicate, AssemblyAI), and bidirectional streams (ElevenLabs/Cartesia/Deepgram WS, realtime). Video has no sync provider at all.
|
||||
2. **Output is never just bytes.** base64, signed URLs with TTLs from 10 minutes (BFL) to 2 days (Veo), URLs that need auth plus redirect (Veo), separate download endpoints (Sora `/content?variant=`), raw bodies (Stability, TTS). Multi-output is the norm.
|
||||
3. **Inputs have roles.** First/last frame, mask, style/subject reference, source video for edit/extend, reference audio, prior generation id, provider-side file handles (`file_id`, `gs://`, `runway://`, `mm_file://`).
|
||||
4. **Partial streaming is modality-specific.** Images: a few whole partial frames. Audio: ordered chunks plus timestamp events. Jobs: status/progress/logs. Video: none.
|
||||
5. **Usage is a union**: tokens, seconds, characters (often only in headers), credits, compute time.
|
||||
6. **Moderation can be partial success** (Veo strips audio but returns video). Deprecations are constant (Sora API shuts down 2026-09-24, Imagen on Gemini API 2026-08-17).
|
||||
|
||||
## Where existing SDKs are weak and we should not be
|
||||
|
||||
- No streaming TTS.
|
||||
- Video handles are experimental start/status pairs; the polling loop lives inside the generate call.
|
||||
- Unsupported inputs become silent warnings arrays, so a request can succeed while dropping your mask.
|
||||
- `n` is fanned out into hidden parallel calls, which obscures cost and idempotency.
|
||||
- Each modality has its own bespoke result type; the file abstraction is a lazy base64/bytes pair with no URL, expiry, or provider ref.
|
||||
- Effect's own `unstable/ai` has no media generation. Nothing in the Effect ecosystem owns this.
|
||||
|
||||
## Design principles
|
||||
|
||||
- **Same shape as LLM.** `X.request(...)` → Schema class; `X.generate(request)` / `X.stream(request)`; `XClient.Service` + `layer`; typed `AIError`. If you know `LLM`, you know `Video`.
|
||||
- **Execution shape is route policy, not API shape.** `Image.generate` returns an image whether the provider is inline or queued. Job control is available uniformly when you want it.
|
||||
- **Errors, not warnings.** Unsupported common fields fail at the protocol boundary with a typed `AIError`, as the LLM routes do today. Provider-side partial results (filtered audio, moderated sample) surface as `notices` on the response, never as silent drops.
|
||||
- **One asset type in, one asset type out**, shared with LLM messages and tool results.
|
||||
- **Typed per-model options**, no hidden fan-out, no implicit retries that spend money.
|
||||
- **Promise API is one mechanism for the whole package**, not a media-only wrapper.
|
||||
- **One construction path per model.** Media models come from per-modality selectors on the configured facade (`openai.image("gpt-image-2")`), the same shape as `openai.responses("gpt-5")`.
|
||||
|
||||
## Public API
|
||||
|
||||
### Model selection
|
||||
|
||||
A model value is built as `OpenAI.configure({ apiKey }).responses("gpt-5")` or `.image("gpt-image-2")`: `configure` fixes credentials, endpoint, and defaults; the selector fixes which of the provider's APIs to hit and binds the typed `providerOptions` generic. Media follows the same shape with one selector per modality — `openai.image(id)` today, `.video(id)` / `.speech(id)` / `.transcription(id)` as those modalities land — mirroring `openai.responses(id)`. `Image.request` accepts `ImageModel` only, exactly as `LLM.request` accepts `LanguageModel`.
|
||||
|
||||
```ts
|
||||
import { OpenAI, Google } from "@opencode/ai/providers"
|
||||
|
||||
const openai = OpenAI.configure({ apiKey }) // OpenAI(...) alone uses env auth (OPENAI_API_KEY)
|
||||
|
||||
LLM.request({ model: openai.responses("gpt-5"), prompt })
|
||||
Image.request({ model: openai.image("gpt-image-2"), prompt })
|
||||
Video.request({ model: google.video("veo-3.1-generate-preview"), prompt })
|
||||
Speech.request({ model: openai.speech("gpt-4o-mini-tts"), text })
|
||||
Transcription.request({ model: openai.transcription("gpt-4o-transcribe"), audio })
|
||||
```
|
||||
|
||||
The request namespace and the selector share one word (`Image.request` + `.image(...)`). That redundancy is accepted: a callable facade returning a lazily resolved ref would be a second way to construct the same model, and the type machinery to infer `providerOptions` through it is not worth one word. Where a provider has two APIs for one modality, the selectors stay explicit (`openai.chat`, a future `google.imagen`), and one default per modality per provider is part of the facade definition (OpenAI image → Images API, Google image → Gemini-native since Imagen on the Gemini API shuts down 2026-08-17). Provider package entrypoints keep `model(modelID, settings)` per modality-specific path, e.g. `@opencode/ai/providers/openai/responses`.
|
||||
|
||||
### `Media` — the asset type
|
||||
|
||||
Replaces `MediaPart.data: string | Uint8Array`, `ImageInput`, `GeneratedImage`, and aligns `Tool.FileContent`.
|
||||
|
||||
```ts
|
||||
import { Media } from "@opencode/ai"
|
||||
|
||||
Media.Source =
|
||||
| { type: "bytes"; data: Uint8Array; mediaType: string }
|
||||
| { type: "base64"; data: string; mediaType: string }
|
||||
| { type: "url"; url: string; mediaType?: string; expiresAt?: number }
|
||||
| { type: "ref"; provider: ProviderID; id: string; mediaType?: string } // file_id, gs://, runway://, prior generation
|
||||
|
||||
class Media.Asset {
|
||||
readonly source: Media.Source
|
||||
readonly mediaType: string // always resolved (sniffed when the provider omits it)
|
||||
readonly kind: "image" | "video" | "audio" | "document" | "other"
|
||||
readonly info?: { width?; height?; durationSeconds?; sampleRate?; channels?; encoding?; format? }
|
||||
readonly expiresAt?: number
|
||||
readonly providerMetadata?: ProviderMetadata
|
||||
readonly headers?: Record<string, string> // transient download credentials (Veo); never in source/JSON
|
||||
|
||||
bytes(): Effect<Uint8Array, AIError, RequestExecutor.Service> // downloads/decodes lazily, cached
|
||||
base64(): Effect<string, AIError, RequestExecutor.Service>
|
||||
dataUrl(): Effect<string, AIError, RequestExecutor.Service>
|
||||
materialize(): Effect<Media.Asset, AIError, RequestExecutor.Service> // url/ref → bytes, before the URL dies
|
||||
}
|
||||
|
||||
Media.bytes(data, mediaType?) Media.base64(data, mediaType?)
|
||||
Media.url(url, options?) Media.ref(provider, id)
|
||||
Media.file(path) // Bun/Node: reads + sniffs; Effect FileSystem variant for layers
|
||||
Media.write(asset, path) // convenience, uses FileSystem
|
||||
```
|
||||
|
||||
Raw-PCM outputs (Gemini TTS, Cartesia raw, Deepgram WS) carry `info.encoding/sampleRate/channels` because there is no container header.
|
||||
|
||||
### Modality namespaces
|
||||
|
||||
Each namespace mirrors `LLM` exactly.
|
||||
|
||||
```ts
|
||||
import { Image, Video, Speech, Transcription } from "@opencode/ai"
|
||||
import { OpenAI, Google, ElevenLabs, Fal } from "@opencode/ai/providers"
|
||||
```
|
||||
|
||||
#### Image
|
||||
|
||||
```ts
|
||||
const request = Image.request({
|
||||
model: openai.image("gpt-image-2"),
|
||||
prompt: "A robot tending a rooftop garden",
|
||||
images: [Media.file("./ref.png")], // references / edit sources
|
||||
mask: Media.file("./mask.png"),
|
||||
n: 2,
|
||||
size: "1536x1024", // or aspectRatio: "3:2"
|
||||
seed: 7,
|
||||
format: "webp",
|
||||
providerOptions: { quality: "high", background: "transparent" }, // typed per model
|
||||
})
|
||||
|
||||
const response = yield* Image.generate(request) // ImageResponse
|
||||
response.image // Media.Asset (first)
|
||||
response.images // Media.Asset[]
|
||||
response.usage // Usage union (see below)
|
||||
response.notices // moderation / partial-result notices
|
||||
|
||||
yield* Image.stream(request) // Stream<ImageEvent>
|
||||
// ImageEvent: generation-queued | generation-progress | image-partial { index, image } | image { index, image } | finish { usage }
|
||||
```
|
||||
|
||||
Editing is not a separate function; `images`/`mask` on the request select the edit path in the route (OpenAI `/images/edits`, Gemini multimodal parts, xAI `/images/edits`). Routes that cannot honor `mask` fail with `Unsupported`.
|
||||
|
||||
#### Video
|
||||
|
||||
Shipped in phase 2 (`src/video.ts`, `src/video-client.ts`, protocols `google-video`, `xai-video`, `fal-video`, `runway-video`).
|
||||
|
||||
```ts
|
||||
const request = Video.request({
|
||||
model: google.video("veo-3.1-generate-preview"),
|
||||
prompt: "Panning wide shot of a calico kitten sleeping in the sunshine",
|
||||
frames: { first: Media.file("./start.png"), last: Media.file("./end.png") },
|
||||
references: [Media.file("./style.png")],
|
||||
video: Media.bytes(previous, "video/mp4"), // edit / extend source
|
||||
durationSeconds: 8,
|
||||
aspectRatio: "16:9",
|
||||
resolution: "1080p",
|
||||
audio: true,
|
||||
n: 1,
|
||||
seed: 7,
|
||||
negativePrompt: "text, watermark", // common, not provider-native
|
||||
providerOptions: { personGeneration: "allow_adult" },
|
||||
})
|
||||
|
||||
// Simple: wait for it.
|
||||
const response = yield* Video.generate(request, { poll: { interval: "10 seconds", timeout: "10 minutes" } })
|
||||
response.video // Media.Asset: url with expiresAt (+ transient `headers` for Veo downloads)
|
||||
response.usage // credits on Runway; the other three report none
|
||||
response.notices // Veo raiMediaFilteredReasons → filtered, xAI respect_moderation → moderated
|
||||
yield* response.video.materialize() // pull bytes before the URL expires
|
||||
|
||||
// Explicit generation control.
|
||||
const generation = yield* Video.start(request) // Generation<VideoResponse>
|
||||
generation.id; generation.status; generation.progress; generation.position; generation.token
|
||||
yield* generation.await({ poll }) // VideoResponse
|
||||
yield* generation.cancel() // fal PUT cancel_url, Runway DELETE /tasks/{id}; no-op for Veo and xAI
|
||||
|
||||
// Resume from another process. The token is validated against the route's codec and refreshed once.
|
||||
const resumed = yield* Video.resume(model, JSON.parse(saved))
|
||||
|
||||
// Progress as a stream.
|
||||
yield* Video.stream(request, { poll }) // Stream<VideoEvent>: generation-queued { id, position } | generation-progress { id, progress } | video { index, video } | finish { usage, notices }
|
||||
```
|
||||
|
||||
Tokens are route-owned JSON: Veo `{ operation }`, xAI `{ requestID }`, Runway `{ taskID }`, fal
|
||||
`{ requestID, statusURL, responseURL, cancelURL }` (fal's follow-up URLs are authoritative and absolute). Common-field
|
||||
lowering per provider: Veo takes inline media only and rejects `audio: false` and `n > 1`; xAI rejects `seed` and
|
||||
`negativePrompt` and routes a `video` input to edits or (`providerOptions.mode: "extend"`) extensions; fal rejects
|
||||
`durationSeconds`, `references`, and `frames.last` because the field names and enums differ per model; Runway passes
|
||||
`aspectRatio` through as its pixel `ratio` and rejects `n`.
|
||||
|
||||
Deferred: `Video.complete(model, token, webhook)` (finish from a webhook payload without polling) and provider poll
|
||||
hints (none of the four providers emit one). Later providers: Luma, Kling, MiniMax, Replicate.
|
||||
|
||||
#### Speech (TTS)
|
||||
|
||||
Shipped in phase 3 (`src/speech.ts`, `src/speech-client.ts`, protocols `openai-speech`, `google-speech`,
|
||||
`elevenlabs-speech`, `cartesia-speech`, `deepgram-speech`; new `ElevenLabs`, `Cartesia`, and `Deepgram` facades).
|
||||
|
||||
```ts
|
||||
const request = Speech.request({
|
||||
model: elevenlabs.speech("eleven_flash_v2_5"),
|
||||
text: "Hello from OpenCode.",
|
||||
voice: "JBFqnCBsd6RMkjVDRZzb", // provider-native identifier, or { id }
|
||||
format: "mp3", // mp3 | wav | pcm | opus | aac | flac | (string & {})
|
||||
speed: 1.0,
|
||||
language: "en",
|
||||
instructions: "Warm, unhurried.", // only OpenAI; elsewhere fails typed
|
||||
timestamps: true, // request alignment; routes without it fail typed
|
||||
providerOptions: { voice_settings: { stability: 0.5 } },
|
||||
})
|
||||
|
||||
const response = yield* Speech.generate(request) // SpeechResponse: audio: Media.Asset, timestamps?, usage?, providerMetadata?
|
||||
yield* Speech.stream(request) // Stream<SpeechEvent>: audio-delta { chunk } | timestamps { items } | finish { audio, usage? }
|
||||
```
|
||||
|
||||
Execution is `MediaProtocol.stream` for every provider: one request whose body is framed and folded by a `step`
|
||||
state machine, with `generate` running the same stream and collecting it. The route submits the request with its
|
||||
`mode` (`"generate" | "stream"`), which lets one provider stay one protocol — OpenAI adds `stream_format: "sse"` (except `tts-1`/`tts-1-hd`, which stream raw bytes), ElevenLabs appends
|
||||
`/stream`, Cartesia switches `/tts/bytes` to `/tts/sse`, Gemini switches `generateContent` to
|
||||
`streamGenerateContent`. The terminal `finish` event carries the assembled asset (every provider's stream is
|
||||
concatenable chunks), so stream consumers also get the whole file and `generate` is just "take `finish`, gather
|
||||
`timestamps`". The cost is memory: a stream holds every chunk until `finish`, so even a consumer that only plays deltas
|
||||
keeps the whole clip in memory. That is bounded by the providers' input text limits (a few minutes of audio); a
|
||||
long-form or session API would need an opt-out.
|
||||
|
||||
**Voice.** `voice?: string | { id: string }`. A string is passed through as the provider's native identifier — a
|
||||
name on OpenAI and Gemini, a voice id on ElevenLabs (path segment) and Cartesia. `{ id }` selects an OpenAI custom
|
||||
voice and is treated as the plain string on routes that do not distinguish custom from built-in. Deepgram's voice is
|
||||
the model id (`aura-2-thalia-en`), so `voice` is `unsupported` there. There is no cross-provider voice catalog or
|
||||
name→id resolution. Multi-speaker (Gemini `speechConfig.multiSpeakerVoiceConfig`) and per-voice settings
|
||||
(ElevenLabs `voice_settings`) go through `providerOptions`.
|
||||
|
||||
**Format and PCM.** `format` is container-level; provider sample rates and bitrates live under `providerOptions`
|
||||
(ElevenLabs `outputFormat`, Cartesia `sampleRate`/`bitRate`/`encoding`, Deepgram `encoding`/`container`/`sampleRate`/
|
||||
`bitRate`). Each protocol maps `format` to its native value (ElevenLabs `mp3_44100_128`/`pcm_24000`/`wav_24000`/
|
||||
`opus_48000_64`, Cartesia `{ container, encoding, sample_rate }`, Deepgram `encoding`+`container`) and declares the
|
||||
asset's media type rather than sniffing, because headerless PCM can look like an MPEG frame sync. Headerless PCM
|
||||
always carries `info.encoding`, `info.sampleRate`, and `info.channels`; its media type is the provider's declaration
|
||||
(Gemini `audio/L16;codec=pcm;rate=24000`, Deepgram's `content-type`) or `audio/pcm`. Gemini returns PCM only, so any
|
||||
other `format` is rejected rather than wrapped as WAV by the route. Every `format` value a route cannot produce (unknown
|
||||
to it, a container on Cartesia SSE, WAV on an ElevenLabs stream, anything but PCM on Gemini) fails the same way as an
|
||||
unsupported field: `UnsupportedOperation` with `operation: "media.format"`.
|
||||
|
||||
**Timestamps.** `timestamps: true` on the request asks for alignment. ElevenLabs selects the `with-timestamps`
|
||||
endpoints (character-level, NDJSON when streaming); Cartesia sets `add_timestamps` on `/tts/sse` (word-level; a
|
||||
`generate` with timestamps collects the SSE stream). OpenAI, Gemini, and Deepgram reject it.
|
||||
|
||||
Common-field lowering per provider:
|
||||
|
||||
| Provider | `voice` | `speed` | `language` | `instructions` | `timestamps` | Usage |
|
||||
|---|---|---|---|---|---|---|
|
||||
| OpenAI | `voice` (name or `{ id }`) | `speed` | unsupported | `instructions` | unsupported | `tokens` from SSE `speech.audio.done` only |
|
||||
| Gemini | `prebuiltVoiceConfig.voiceName` | unsupported | `speechConfig.languageCode` | unsupported (direct in text) | unsupported | `tokens` from `usageMetadata` |
|
||||
| ElevenLabs | path voice id (required) | `voice_settings.speed` | `language_code` | unsupported | `with-timestamps` | `credits` from `character-cost` header |
|
||||
| Cartesia | `voice` (required) | `generation_config.speed` | `language` | unsupported | `add_timestamps` | none |
|
||||
| Deepgram | unsupported (voice is the model) | `speed` query | unsupported | unsupported | unsupported | `characters` from `dg-char-count` header |
|
||||
|
||||
Deferred: `Speech.session(...)` — input-streaming TTS where text arrives incrementally over a WebSocket (ElevenLabs
|
||||
`stream-input`, Cartesia WebSocket contexts, Deepgram WebSocket speak) — is a separate scoped resource, not part of
|
||||
`generate`/`stream`, and ships with the realtime work in phase 5.
|
||||
|
||||
#### Transcription (STT)
|
||||
|
||||
Shipped as the second half of phase 3 (`src/transcription.ts`, `src/transcription-client.ts`, protocols
|
||||
`openai-transcription`, `google-transcription`, `deepgram-transcription`, `assemblyai-transcription`; new `AssemblyAI`
|
||||
facade).
|
||||
|
||||
```ts
|
||||
const request = Transcription.request({
|
||||
model: openai.transcription("gpt-4o-transcribe-diarize"),
|
||||
audio: yield* Media.file("./call.wav"),
|
||||
language: "en", // provider-native passthrough
|
||||
timestamps: "segment", // none | segment | word
|
||||
diarize: true,
|
||||
speakers: 2, // expected count, hint only (AssemblyAI)
|
||||
providerOptions: { known_speaker_names: ["agent"] },
|
||||
})
|
||||
|
||||
const response = yield* Transcription.generate(request)
|
||||
response.text; response.segments; response.words; response.language; response.durationSeconds; response.usage
|
||||
yield* Transcription.stream(request) // Stream<TranscriptionEvent>: generation-queued | generation-progress | text-delta | segment | finish
|
||||
const generation = yield* Transcription.start(request) // queued routes only
|
||||
yield* Transcription.resume(model, token)
|
||||
```
|
||||
|
||||
Transcription is the first modality whose providers span all three protocol kinds, and it needed no fourth kind.
|
||||
Every `MediaRoute` now carries its `kind`; `TranscriptionRoute` is the union of the inline, stream, and queued routes;
|
||||
`TranscriptionModel.fromRoute` is overloaded per protocol kind (arity picks the overload: `<Options>`,
|
||||
`<Options, Frame, State>`, `<Options, Token>`) and composes through `MediaRoute.inline` / `stream` / `queued`; and
|
||||
`TranscriptionClient` dispatches on `route.kind`. `generate` on a queued route is `start` then `await`; `stream` on an
|
||||
inline route is the response as a single `finish`, and on a queued route it is the status observations followed by
|
||||
`finish`. `start` / `resume` on a non-queued route fail with `UnsupportedOperation` (`transcription.start`). The
|
||||
`finish` event carries the whole transcript (text, segments, words, language, duration, usage), so the stream route's
|
||||
`collect` is just "take `finish`".
|
||||
|
||||
The route layer gained a `binary` body with array-valued `query` (Deepgram) and `Queued.start.prepare` (AssemblyAI's
|
||||
upload); `packages/ai/AGENTS.md` (Media Routes) describes both.
|
||||
|
||||
Settled rules:
|
||||
|
||||
- **Timestamps.** A granularity the selected route or model cannot produce fails as `UnsupportedOperation`
|
||||
(`media.timestamps`), following Speech; a route that returns more than asked (Deepgram and AssemblyAI always return
|
||||
words) is not stripped. Segments always carry start and end times: Gemini times each transcription part from its
|
||||
word offsets, so segment timestamps and diarization also request word offsets there.
|
||||
- **Diarization.** `diarize` means segments (and words, where the provider labels them) carry `speaker`. Labels are
|
||||
provider-native strings — OpenAI `A` or a known speaker name, Deepgram `0`, Gemini `spk:0`, AssemblyAI `A` — with no
|
||||
cross-provider speaker model. `speakers` is a hint; only AssemblyAI (`speakers_expected`) accepts it.
|
||||
- **Language** is passed through (`language`, OpenAI `gpt-transcribe` `languages[]`, Gemini `languageCodes`,
|
||||
AssemblyAI `language_code`). `response.language` is the provider's own value, lowercased but not normalized: an
|
||||
ISO code on most routes, `english` from whisper-1, `en_us` from AssemblyAI. Deepgram and AssemblyAI assume English
|
||||
unless asked to detect, so a missing `language` enables their detection.
|
||||
- **Gemini** requires a transcribe model; other model ids fail with `UnsupportedOperation` before the call, because
|
||||
general models ignore `audioTranscriptionConfig` and answer conversationally. Streamed chunks carry whole speaker
|
||||
turns (one part per turn), which join with a space.
|
||||
- **Streaming inline providers** emit only `finish`; deltas are never faked.
|
||||
- **Units.** AssemblyAI milliseconds and Gemini protobuf durations (`"0.400s"`) are normalized to seconds at the
|
||||
protocol boundary.
|
||||
|
||||
| Provider | Kind | Audio input | `timestamps` | `diarize` | Unsupported | Usage |
|
||||
|---|---|---|---|---|---|---|
|
||||
| OpenAI | stream (`stream: true` in `stream` mode) | multipart `file` (inline only) | `whisper-1` (`verbose_json`); diarize model: `segment` | `gpt-4o-transcribe-diarize` (`diarized_json`) | `speakers`; `prompt` on the diarize model; streaming on `whisper-1` | `tokens` or `seconds` |
|
||||
| Gemini | stream (`generateContent` / `streamGenerateContent`) | `inlineData` or Gemini Files `fileData` | `audioTranscriptionConfig.wordTimestamp` | `audioTranscriptionConfig.diarization` | `prompt`, `speakers` | `tokens` |
|
||||
| Deepgram | inline | raw body, or JSON `{ url }` | words always; `segment` → `utterances` | `diarize_model=latest` + `utterances` | `prompt`, `speakers` | `seconds` (`metadata.duration`) |
|
||||
| AssemblyAI | queued (upload → submit → poll) | `/v2/upload` then `audio_url`, or a URL | words always; `segment` → `speaker_labels` | `speaker_labels` | — | `seconds` (`audio_duration`) |
|
||||
|
||||
Deferred: `Transcription.session(...)` — realtime STT over WebSocket (Deepgram live, AssemblyAI streaming, ElevenLabs
|
||||
realtime, OpenAI realtime transcription) — is the same future scoped `session` shape as input-streaming TTS and ships
|
||||
with the realtime work in phase 5. ElevenLabs Scribe is not implemented yet.
|
||||
|
||||
### `Generation` — shared async execution
|
||||
|
||||
```ts
|
||||
class Generation<Response> {
|
||||
readonly id: string
|
||||
readonly route: GenerationRoute<Response> // token-free: { status, result, cancel?: Effect; pollHint? } closed over the decoded token
|
||||
readonly token: unknown // route-owned serializable JSON
|
||||
readonly status: "queued" | "running" | "completed" | "failed" | "cancelled" | "expired"
|
||||
readonly progress?: number // 0..1, normalized
|
||||
readonly position?: number
|
||||
readonly expiresAt?: number
|
||||
refresh(): Effect<Generation<Response>, AIError>
|
||||
result(): Effect<Response, AIError>
|
||||
await(options?: AwaitOptions): Effect<Response, AIError>
|
||||
cancel(): Effect<void, AIError>
|
||||
events(options?: AwaitOptions): Stream<GenerationEvent, AIError> // fails with Timeout past poll.timeout, checked per observation
|
||||
}
|
||||
|
||||
AwaitOptions = { poll?: Poll }
|
||||
Poll = { interval?: Duration; timeout?: Duration; schedule?: Schedule } // route may override from provider hints (`openai-poll-after-ms`)
|
||||
```
|
||||
|
||||
`Generation` is not video-specific. Image routes on BFL, fal, and Replicate are queued; `Image.start` exists for them. A route declares itself `inline` or `queued`; `generate` on a queued route is `start` then `await`.
|
||||
|
||||
### Usage
|
||||
|
||||
```ts
|
||||
Usage =
|
||||
| { type: "tokens"; input; output; total; details? }
|
||||
| { type: "seconds"; seconds }
|
||||
| { type: "characters"; characters }
|
||||
| { type: "credits"; credits }
|
||||
| { type: "compute"; seconds }
|
||||
```
|
||||
|
||||
Header-only usage (ElevenLabs `character-cost`, Deepgram `dg-char-count`) is lifted into `usage` by the route.
|
||||
|
||||
### Promise API — `@opencode/ai/promise`
|
||||
|
||||
Mirrors the `packages/plugin/src/effect` and `packages/plugin/src/promise` split that already exists in this repo. One mechanism for LLM and media.
|
||||
|
||||
```ts
|
||||
import { AI } from "@opencode/ai/promise"
|
||||
|
||||
const ai = AI.make() // ManagedRuntime over RequestExecutor.fetchLayer + all clients
|
||||
// AI.make({ layer }) to inject a custom executor / recorder / middleware
|
||||
|
||||
const image = await ai.image.generate({ model, prompt })
|
||||
await image.image.bytes()
|
||||
|
||||
for await (const event of ai.speech.stream({ model, text, voice })) { … }
|
||||
|
||||
const generation = await ai.video.start({ model, prompt })
|
||||
const video = await generation.await({ poll: { interval: 10_000 }, signal })
|
||||
const resumed = ai.video.resume(model, JSON.parse(saved))
|
||||
|
||||
const text = await ai.llm.generate({ model, prompt }) // closes today's gap: LLM has no promise API either
|
||||
for await (const event of ai.llm.stream(request)) { … }
|
||||
|
||||
await ai.dispose()
|
||||
```
|
||||
|
||||
Streams become `AsyncIterable` via `Stream.toAsyncIterable`. `AIError` is thrown as-is. `AbortSignal` maps to interruption. Nothing in `src/*` except this entrypoint knows about promises.
|
||||
|
||||
### Providers
|
||||
|
||||
Existing facades gain per-modality selectors; the modality routes each facade provides:
|
||||
|
||||
| Facade | llm | image | video | speech | transcription | other |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `OpenAI` | responses (default), chat | Images API | Sora (deprecated 2026-09-24) | ✓ | ✓ | |
|
||||
| `Google` | Gemini | Gemini-native (default), `imagen` | Veo | Gemini TTS | `gemini-3.5-transcribe` | |
|
||||
| `XAI` | ✓ | ✓ | ✓ | | | |
|
||||
| `ElevenLabs` | | | | ✓ | Scribe | soundEffect, music |
|
||||
| `Cartesia` | | | | ✓ | | |
|
||||
| `Deepgram` | | | | Aura | ✓ | |
|
||||
| `Fal` | | ✓ | ✓ | | | |
|
||||
| `AssemblyAI` | | | | | ✓ (queued) | |
|
||||
| `Replicate`, `Runway`, `Luma`, `Kling`, `MiniMax`, `BlackForestLabs`, `Stability` | | per provider | | | | |
|
||||
|
||||
New facades follow the existing one-file-per-provider rule. Package entrypoints are modality-specific, such as `@opencode/ai/providers/openai/images`, and return the concrete model.
|
||||
|
||||
`ImageModel<Options>` already gives typed `providerOptions` per model; `VideoModel`, `SpeechModel`, `TranscriptionModel` follow the same generic. A shared `MediaModel` union is what `Generation` and the promise client key on.
|
||||
|
||||
### Routes and protocols
|
||||
|
||||
Media does not fit the LLM four-axis route (SSE frames → event state machine) except for streaming TTS/STT. Reuse `Endpoint`, `Auth`, `Framing`, `RequestExecutor`, and add media protocol kinds:
|
||||
|
||||
- `MediaProtocol.inline` — `body.from(request)` (JSON, multipart, or query), `response.decode(response)` (JSON, or binary body → `Media.Asset`).
|
||||
- `MediaProtocol.queued` — `start` (body + decode to `{ token, snapshot }`), `status`, `result`, optional `cancel`, `pollHint`, and a `token` codec. `result` is always a separate GET (against the status document for Veo/xAI/Runway, fal's `response_url` otherwise) so `await` after `start` and after `resume` share one path. `PollContext.auth` hands the auth headers the route sent to the protocol for output URLs that need them (Veo downloads); they become transient `Media.Asset.headers`, never part of `source`. There is no separate `download` step: `Media.Asset.bytes()` downloads through the executor with those headers. `MediaRoute.inline(...)` / `MediaRoute.queued(...)` compose each kind with endpoint and auth; the queued route decodes the token once and hands `Generation` a token-free `{ status, result, cancel? }`.
|
||||
- `MediaProtocol.stream` — `body.from(request)` over the request plus its `mode`, `frames` (a function that picks the framing for the call: `Framing.sse`, `lines`, `document`, or the raw bytes), fresh per-response `initial()` state, `step` emitting modality events, and `finish(state, context)` — with the observed response for header-only usage — emitting exactly one terminal event or failing as an incomplete stream. The route fills `reason.http` on stream errors. `MediaRoute.stream(...)` exposes `stream` and `generate` (the same stream folded by the modality's `collect`).
|
||||
|
||||
`MediaRoute.inline` / `MediaRoute.queued` / `MediaRoute.stream` compose one protocol kind with endpoint/auth and tag the route with its `kind`; `ImageModel`/`VideoModel`/`SpeechModel`/`TranscriptionModel` share the `MediaModel` base (`src/media-model.ts`).
|
||||
|
||||
### LLM integration
|
||||
|
||||
- `MediaPart` becomes `{ type: "media"; media: Media.Asset; … }` so protocols branch on `kind` and can pass `url`/`ref` sources through natively (OpenAI `image_url`, Gemini `fileData`).
|
||||
- New `LLMEvent`s: `media { media: Media.Asset }` so Gemini inline image output is first-class instead of dropped. OpenAI Responses `image_generation_call` keeps its single carrier — the provider-executed `tool-result` with `file` content — because Core consumes hosted tool-result content today and has no `media` event handling yet; it switches to the `media` carrier when Core adopts the event, so the image is never emitted twice.
|
||||
- `Message.assistant([...])` accepts media parts; Gemini multi-turn image editing replays them.
|
||||
- `Tool.FileContent` aligns with `Media.Source`.
|
||||
|
||||
## Decisions
|
||||
|
||||
All settled:
|
||||
|
||||
1. **Per-modality selectors** (`openai.image(id)`, `.video`, `.speech`, `.transcription`) name media models, mirroring `openai.responses(id)`. The one-word overlap with the request namespace is accepted over a callable-facade `ModelRef` as a second construction path.
|
||||
2. **`providerOptions` everywhere** (rename current `Image.options`) for consistency with LLM.
|
||||
3. **No hidden `n` fan-out.** `n` lowers natively; routes that cannot do `n > 1` fail typed. Callers use `Effect.all` / `Promise.all` explicitly.
|
||||
4. **Errors over warnings** for unsupported common fields; `notices` for provider-side partial results only.
|
||||
5. **`Media.Asset` is a class** (lazy bytes, cached) with `Media.Source` as the serializable Schema for wire/persistence. `Asset.from(source)` / `asset.source` round-trip losslessly. Same pattern as `LanguageModel` today.
|
||||
6. **Promise entrypoint**: `@opencode/ai/promise` exporting `AI.make(options?: { layer? })` plus a module-level default `ai` for scripts, covering LLM too.
|
||||
7. **Modality set for v1**: `Image`, `Video`, `Speech`, `Transcription`. `Music`/`SoundEffect` and `session` (bidirectional WS, realtime) are designed-for but deferred.
|
||||
8. **Sora is skipped** (API shuts down 2026-09-24). Video launches with Veo, xAI, fal, Runway.
|
||||
|
||||
## Build order
|
||||
|
||||
Foundation + Image ship together as the reference implementation, serially. Video, Speech, and Transcription then proceed in parallel on separate branches. Image jobs and partial streaming come last, after Video has hardened `Generation`.
|
||||
|
||||
## Phasing
|
||||
|
||||
1. **Foundation** — per-modality selectors, `Media`, `Generation`, `Poll`, `Usage` union, `MediaProtocol` kinds, `@opencode/ai/promise` with `llm` + `image`. Port the five existing image protocols onto it. Unify `MediaPart` and add the `media` LLM event (fixes Gemini image output being dropped).
|
||||
2. **Video** — ✅ Veo, xAI, fal, Runway shipped (`MediaProtocol.queued`, `Video.start/generate/resume/stream`, promise `ai.video`). Deferred: `Video.complete` (webhooks), Luma, Kling, MiniMax, Replicate.
|
||||
3. **Speech + Transcription** — ✅ Speech: OpenAI, Gemini TTS, ElevenLabs, Cartesia, Deepgram shipped (`MediaProtocol.stream`, `Speech.generate/stream`, promise `ai.speech`). ✅ Transcription: OpenAI, Gemini, Deepgram, AssemblyAI shipped across all three route kinds (`Transcription.generate/stream/start/resume`, promise `ai.transcription`). Pending: ElevenLabs Scribe. Deferred: `Speech.session` and `Transcription.session` (WebSocket streaming).
|
||||
4. **Image queued routes and partials** — BFL, fal, Replicate, Stability; OpenAI `partial_images` streaming.
|
||||
5. **Later** — ElevenLabs music/SFX, Lyria, `Speech.session` / `Transcription.session`, realtime.
|
||||
|
||||
Core adoption (session attachments beyond png/jpeg/gif/webp/pdf, image-generation tool, TUI rendering) comes after phase 1 and is a Core concern.
|
||||
@@ -1,5 +1,17 @@
|
||||
import { Config, Effect, Formatter, Layer, Schema, Stream } from "effect"
|
||||
import { LLM, LLMClient, LLMRequest, Message, ProviderID, Tool, ToolRuntime } from "@opencode/ai"
|
||||
import { NodeFileSystem } from "@effect/platform-node"
|
||||
import {
|
||||
Image,
|
||||
ImageClient,
|
||||
LLM,
|
||||
LLMClient,
|
||||
LLMRequest,
|
||||
Media,
|
||||
Message,
|
||||
ProviderID,
|
||||
Tool,
|
||||
ToolRuntime,
|
||||
} from "@opencode/ai"
|
||||
import { Route, Auth, Endpoint, Framing, Protocol, RequestExecutor } from "@opencode/ai/route"
|
||||
import { OpenAI } from "@opencode/ai/providers"
|
||||
|
||||
@@ -16,15 +28,18 @@ import { OpenAI } from "@opencode/ai/providers"
|
||||
|
||||
const apiKey = Config.redacted("OPENAI_API_KEY")
|
||||
|
||||
// 1. Pick a model. The provider helper records provider identity, protocol
|
||||
// choice, capabilities, deployment options, authentication, and defaults.
|
||||
const model = OpenAI.configure({
|
||||
// 1. Configure a provider. The configured facade records provider identity,
|
||||
// deployment options, authentication, and defaults. Per-modality selectors pick
|
||||
// the API: `.responses(...)` / `.chat(...)` for LLM calls and `.image(...)` for
|
||||
// image generation.
|
||||
const openai = OpenAI.configure({
|
||||
apiKey,
|
||||
generation: { maxTokens: 160 },
|
||||
providerOptions: {
|
||||
store: false,
|
||||
},
|
||||
}).model("gpt-4o-mini")
|
||||
})
|
||||
const model = openai.responses("gpt-4o-mini")
|
||||
|
||||
// 2. Build a provider-neutral request. This is useful when reusing one request
|
||||
// across generate and stream examples.
|
||||
@@ -209,18 +224,39 @@ const FakeEcho = {
|
||||
}),
|
||||
}
|
||||
|
||||
// 8. Image generation uses the same facade and the same request/generate shape.
|
||||
// `response.image` is a `Media.Asset`: bytes decode lazily and are cached, and
|
||||
// `Media.write` persists them through the Effect `FileSystem`.
|
||||
const generateImage = Effect.gen(function* () {
|
||||
const response = yield* Image.generate({
|
||||
model: openai.image("gpt-image-1-mini"),
|
||||
prompt: "A flat black circle centered on a plain white background.",
|
||||
size: "1024x1024",
|
||||
format: "jpeg",
|
||||
providerOptions: { quality: "low" },
|
||||
})
|
||||
|
||||
console.log("\n== image ==")
|
||||
console.log("media type:", response.image.mediaType)
|
||||
console.log("bytes:", (yield* response.image.bytes()).byteLength)
|
||||
console.log("usage", Formatter.formatJson(response.usage, { space: 2 }))
|
||||
yield* Media.write(response.image, "tutorial-image.jpg").pipe(Effect.provide(NodeFileSystem.layer))
|
||||
})
|
||||
|
||||
// Provide the LLM runtime and the HTTP request executor once. Keep one path
|
||||
// enabled at a time so the tutorial can demonstrate generate, stream, or
|
||||
// tool-loop behavior without spending tokens on every example.
|
||||
const requestExecutorLayer = RequestExecutor.fetchLayer
|
||||
const llmClientLayer = LLMClient.layer.pipe(Layer.provide(requestExecutorLayer))
|
||||
const imageClientLayer = ImageClient.layer.pipe(Layer.provide(requestExecutorLayer))
|
||||
|
||||
const program = Effect.gen(function* () {
|
||||
// yield* generateOnce
|
||||
// yield* streamText
|
||||
// yield* generateStructuredObject
|
||||
// yield* generateDynamicObject.pipe(Effect.andThen((response) => Effect.sync(() => console.log(response.object))))
|
||||
// yield* generateImage
|
||||
yield* streamWithTools
|
||||
}).pipe(Effect.provide(Layer.mergeAll(requestExecutorLayer, llmClientLayer)))
|
||||
}).pipe(Effect.provide(Layer.mergeAll(requestExecutorLayer, llmClientLayer, imageClientLayer)))
|
||||
|
||||
Effect.runPromise(program)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"$schema": "https://json.schemastore.org/package.json",
|
||||
"version": "2.0.8",
|
||||
"version": "2.0.15",
|
||||
"name": "@opencode/ai",
|
||||
"type": "module",
|
||||
"license": "MIT",
|
||||
@@ -15,6 +15,7 @@
|
||||
],
|
||||
"exports": {
|
||||
".": "./src/index.ts",
|
||||
"./promise": "./src/promise.ts",
|
||||
"./testing": "./src/testing.ts",
|
||||
"./*": "./src/*.ts"
|
||||
},
|
||||
|
||||
@@ -104,6 +104,83 @@ const PROVIDERS: ReadonlyArray<Provider> = [
|
||||
vars: [{ name: "XAI_API_KEY" }],
|
||||
validate: (env) => validateBearer("https://api.x.ai/v1/models", Redacted.make(env.XAI_API_KEY)),
|
||||
},
|
||||
{
|
||||
id: "fal",
|
||||
label: "fal",
|
||||
tier: "canary",
|
||||
note: "fal queue video recorded tests",
|
||||
vars: [{ name: "FAL_KEY" }],
|
||||
// fal has no free authenticated list endpoint; a 404 for an unknown request id proves the key was accepted.
|
||||
validate: (env) =>
|
||||
Effect.gen(function* () {
|
||||
const http = yield* HttpClient.HttpClient
|
||||
const response = yield* http.execute(
|
||||
HttpClientRequest.get(
|
||||
"https://queue.fal.run/fal-ai/veo3.1/requests/00000000-0000-0000-0000-000000000000/status",
|
||||
).pipe(HttpClientRequest.setHeaders({ authorization: `Key ${Redacted.value(Redacted.make(env.FAL_KEY))}` })),
|
||||
)
|
||||
if (response.status === 404) return undefined
|
||||
return yield* responseError(response)
|
||||
}),
|
||||
},
|
||||
{
|
||||
id: "runway",
|
||||
label: "Runway",
|
||||
tier: "canary",
|
||||
note: "Runway task video recorded tests",
|
||||
vars: [{ name: "RUNWAYML_API_SECRET" }],
|
||||
validate: (env) =>
|
||||
validateBearer("https://api.dev.runwayml.com/v1/organization", Redacted.make(env.RUNWAYML_API_SECRET), {
|
||||
"X-Runway-Version": "2024-11-06",
|
||||
}),
|
||||
},
|
||||
{
|
||||
id: "elevenlabs",
|
||||
label: "ElevenLabs",
|
||||
tier: "canary",
|
||||
note: "ElevenLabs text-to-speech recorded tests",
|
||||
vars: [{ name: "ELEVENLABS_API_KEY" }],
|
||||
validate: (env) =>
|
||||
HttpClientRequest.get("https://api.elevenlabs.io/v1/models").pipe(
|
||||
HttpClientRequest.setHeader("xi-api-key", Redacted.value(Redacted.make(env.ELEVENLABS_API_KEY))),
|
||||
executeRequest,
|
||||
),
|
||||
},
|
||||
{
|
||||
id: "cartesia",
|
||||
label: "Cartesia",
|
||||
tier: "canary",
|
||||
note: "Cartesia text-to-speech recorded tests",
|
||||
vars: [{ name: "CARTESIA_API_KEY" }],
|
||||
validate: (env) =>
|
||||
validateBearer("https://api.cartesia.ai/voices?limit=1", Redacted.make(env.CARTESIA_API_KEY), {
|
||||
"Cartesia-Version": "2026-08-14",
|
||||
}),
|
||||
},
|
||||
{
|
||||
id: "deepgram",
|
||||
label: "Deepgram",
|
||||
tier: "canary",
|
||||
note: "Deepgram Aura text-to-speech and Nova transcription recorded tests",
|
||||
vars: [{ name: "DEEPGRAM_API_KEY" }],
|
||||
validate: (env) =>
|
||||
HttpClientRequest.get("https://api.deepgram.com/v1/projects").pipe(
|
||||
HttpClientRequest.setHeader("authorization", `Token ${Redacted.value(Redacted.make(env.DEEPGRAM_API_KEY))}`),
|
||||
executeRequest,
|
||||
),
|
||||
},
|
||||
{
|
||||
id: "assemblyai",
|
||||
label: "AssemblyAI",
|
||||
tier: "canary",
|
||||
note: "AssemblyAI queued transcription recorded tests",
|
||||
vars: [{ name: "ASSEMBLYAI_API_KEY" }],
|
||||
validate: (env) =>
|
||||
HttpClientRequest.get("https://api.assemblyai.com/v2/transcript?limit=1").pipe(
|
||||
HttpClientRequest.setHeader("authorization", Redacted.value(Redacted.make(env.ASSEMBLYAI_API_KEY))),
|
||||
executeRequest,
|
||||
),
|
||||
},
|
||||
{
|
||||
id: "cloudflare-ai-gateway",
|
||||
label: "Cloudflare AI Gateway",
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
export { EvaluationClient } from "./experimental/evaluation-client.js"
|
||||
export {
|
||||
BooleanAnswer,
|
||||
BooleanQuestion,
|
||||
ChoiceAnswer,
|
||||
ChoiceQuestion,
|
||||
Evaluation,
|
||||
EvaluationAnswer,
|
||||
EvaluationInput,
|
||||
EvaluationModel,
|
||||
EvaluationModelSchema,
|
||||
EvaluationQuestion,
|
||||
EvaluationRequest,
|
||||
EvaluationResponse,
|
||||
EvaluationRounding,
|
||||
ScoreAnswer,
|
||||
ScoreQuestion,
|
||||
} from "./experimental/evaluation.js"
|
||||
export type {
|
||||
AnswerFor,
|
||||
AnswersFor,
|
||||
EvaluationModelOptions,
|
||||
EvaluationOptions,
|
||||
EvaluationQuestions,
|
||||
EvaluationRequestFor,
|
||||
EvaluationRequestInput,
|
||||
EvaluationResponseFor,
|
||||
EvaluationRoute,
|
||||
} from "./experimental/evaluation.js"
|
||||
@@ -0,0 +1,96 @@
|
||||
import { Context, Effect, Layer } from "effect"
|
||||
import { RequestExecutor } from "../route/executor.js"
|
||||
import { AIError, InvalidProviderOutputError, mergeHttpOptions } from "../schema/index.js"
|
||||
import { sanitizeSurrogates } from "../utils/sanitize.js"
|
||||
import {
|
||||
type EvaluationOptions,
|
||||
type EvaluationQuestions,
|
||||
type EvaluationRequestFor,
|
||||
type EvaluationResponseFor,
|
||||
} from "./evaluation.js"
|
||||
|
||||
export type Execute = RequestExecutor.Interface["execute"]
|
||||
|
||||
export interface Interface {
|
||||
readonly evaluate: <Options extends EvaluationOptions, const Questions extends EvaluationQuestions>(
|
||||
request: EvaluationRequestFor<Options, Questions>,
|
||||
) => Effect.Effect<EvaluationResponseFor<Questions>, AIError>
|
||||
}
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/AI/Experimental/EvaluationClient") {}
|
||||
|
||||
export const evaluate = <Options extends EvaluationOptions, const Questions extends EvaluationQuestions>(
|
||||
request: EvaluationRequestFor<Options, Questions>,
|
||||
): Effect.Effect<EvaluationResponseFor<Questions>, AIError, Service> =>
|
||||
Effect.flatMap(Service, (client) => client.evaluate(request))
|
||||
|
||||
export const layer: Layer.Layer<Service, never, RequestExecutor.Service> = Layer.effect(
|
||||
Service,
|
||||
Effect.gen(function* () {
|
||||
const executor = yield* RequestExecutor.Service
|
||||
return Service.of({
|
||||
evaluate: (request) =>
|
||||
request.model.route
|
||||
.evaluate(
|
||||
{
|
||||
...sanitizeSurrogates({
|
||||
...request,
|
||||
model: undefined,
|
||||
http: mergeHttpOptions(request.model.http, request.http),
|
||||
}),
|
||||
model: request.model,
|
||||
},
|
||||
executor.execute,
|
||||
)
|
||||
.pipe(
|
||||
Effect.flatMap((response) => {
|
||||
const questions = Object.entries(request.questions)
|
||||
if (
|
||||
questions.length === Object.keys(response.answers).length &&
|
||||
questions.every(([id, question]) => {
|
||||
const answer = response.answers[id]
|
||||
if (question.type === "boolean") return answer?.type === "boolean"
|
||||
if (question.type === "choice") {
|
||||
if (answer?.type !== "choice" || !Object.hasOwn(question.criteria, answer.choice)) return false
|
||||
if (answer.probabilities === undefined) return true
|
||||
const keys = Object.keys(question.criteria)
|
||||
const probabilities = answer.probabilities
|
||||
return (
|
||||
Object.keys(probabilities).length === keys.length &&
|
||||
keys.every((key) => Object.hasOwn(probabilities, key))
|
||||
)
|
||||
}
|
||||
if (answer?.type !== "score" || answer.score < 0 || answer.score > question.criteria.length - 1)
|
||||
return false
|
||||
if (answer.probabilities === undefined) return true
|
||||
const keys = question.criteria.map((_, index) => String(index))
|
||||
const probabilities = answer.probabilities
|
||||
return (
|
||||
Object.keys(probabilities).length === keys.length &&
|
||||
keys.every((key) => Object.hasOwn(probabilities, key))
|
||||
)
|
||||
})
|
||||
)
|
||||
return Effect.succeed(response as EvaluationResponseFor<typeof request.questions>)
|
||||
return Effect.fail(
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({
|
||||
route: request.model.route.id,
|
||||
message: "Evaluation answers do not match the requested questions",
|
||||
cause: response.answers,
|
||||
}),
|
||||
}),
|
||||
)
|
||||
}),
|
||||
),
|
||||
})
|
||||
}),
|
||||
)
|
||||
export const fetchLayer = layer.pipe(Layer.provide(RequestExecutor.fetchLayer))
|
||||
|
||||
export const EvaluationClient = {
|
||||
Service,
|
||||
layer,
|
||||
fetchLayer,
|
||||
evaluate,
|
||||
} as const
|
||||
@@ -0,0 +1,245 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import {
|
||||
AIError,
|
||||
HttpOptions,
|
||||
InvalidRequestError,
|
||||
ModelID,
|
||||
ProviderID,
|
||||
ProviderMetadata,
|
||||
Usage,
|
||||
} from "../schema/index.js"
|
||||
import { EvaluationClient, Service, type Execute } from "./evaluation-client.js"
|
||||
|
||||
export const EvaluationInput = Schema.Union([Schema.String, Schema.JsonObject, Schema.Array(Schema.Json)])
|
||||
export type EvaluationInput = Schema.Schema.Type<typeof EvaluationInput>
|
||||
|
||||
const EvaluationCriterion = Schema.NullOr(EvaluationInput)
|
||||
const ChoiceCriteria = Schema.Record(Schema.String, EvaluationCriterion).pipe(
|
||||
Schema.refine((x): x is typeof x => Object.keys(x).length > 0, {
|
||||
message: "Choice criteria must be a nonempty option map",
|
||||
}),
|
||||
)
|
||||
|
||||
export const ChoiceQuestion = Schema.Struct({
|
||||
type: Schema.Literal("choice"),
|
||||
instructions: EvaluationInput,
|
||||
criteria: ChoiceCriteria,
|
||||
})
|
||||
export type ChoiceQuestion = Schema.Schema.Type<typeof ChoiceQuestion>
|
||||
|
||||
export const ScoreQuestion = Schema.Struct({
|
||||
type: Schema.Literal("score"),
|
||||
instructions: EvaluationInput,
|
||||
criteria: Schema.Array(EvaluationCriterion).check(Schema.isMinLength(2)),
|
||||
})
|
||||
export type ScoreQuestion = Schema.Schema.Type<typeof ScoreQuestion>
|
||||
|
||||
export const BooleanQuestion = Schema.Struct({
|
||||
type: Schema.Literal("boolean"),
|
||||
instructions: EvaluationInput,
|
||||
criteria: Schema.optional(
|
||||
Schema.Struct({
|
||||
true: Schema.optional(EvaluationCriterion),
|
||||
false: Schema.optional(EvaluationCriterion),
|
||||
}),
|
||||
),
|
||||
})
|
||||
export type BooleanQuestion = Schema.Schema.Type<typeof BooleanQuestion>
|
||||
|
||||
export const EvaluationQuestion = Schema.Union([ChoiceQuestion, ScoreQuestion, BooleanQuestion]).pipe(
|
||||
Schema.toTaggedUnion("type"),
|
||||
)
|
||||
export type EvaluationQuestion = Schema.Schema.Type<typeof EvaluationQuestion>
|
||||
export type EvaluationQuestions = Readonly<Record<string, EvaluationQuestion>>
|
||||
const EvaluationQuestions = Schema.Record(Schema.String, EvaluationQuestion).pipe(
|
||||
Schema.refine((x): x is typeof x => Object.keys(x).length > 0, {
|
||||
message: "Evaluation questions must be a nonempty map",
|
||||
}),
|
||||
)
|
||||
|
||||
const Probability = Schema.Number.check(Schema.isBetween({ minimum: 0, maximum: 1 }))
|
||||
|
||||
export const ChoiceAnswer = Schema.Struct({
|
||||
type: Schema.Literal("choice"),
|
||||
choice: Schema.String,
|
||||
probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
|
||||
})
|
||||
export type ChoiceAnswer = Schema.Schema.Type<typeof ChoiceAnswer>
|
||||
|
||||
export const ScoreAnswer = Schema.Struct({
|
||||
type: Schema.Literal("score"),
|
||||
score: Schema.Number,
|
||||
probabilities: Schema.optional(Schema.Record(Schema.String, Probability)),
|
||||
})
|
||||
export type ScoreAnswer = Schema.Schema.Type<typeof ScoreAnswer>
|
||||
|
||||
export const BooleanAnswer = Schema.Struct({
|
||||
type: Schema.Literal("boolean"),
|
||||
probability: Probability,
|
||||
})
|
||||
export type BooleanAnswer = Schema.Schema.Type<typeof BooleanAnswer>
|
||||
|
||||
export const EvaluationAnswer = Schema.Union([ChoiceAnswer, ScoreAnswer, BooleanAnswer]).pipe(
|
||||
Schema.toTaggedUnion("type"),
|
||||
)
|
||||
export type EvaluationAnswer = Schema.Schema.Type<typeof EvaluationAnswer>
|
||||
|
||||
export type AnswerFor<Question extends EvaluationQuestion> = Question extends {
|
||||
readonly type: "choice"
|
||||
readonly criteria: infer Criteria
|
||||
}
|
||||
? {
|
||||
readonly type: "choice"
|
||||
readonly choice: Extract<keyof Criteria, string>
|
||||
readonly probabilities?: Readonly<Record<Extract<keyof Criteria, string>, number>>
|
||||
}
|
||||
: Question extends { readonly type: "score" }
|
||||
? ScoreAnswer
|
||||
: BooleanAnswer
|
||||
|
||||
export type AnswersFor<Questions extends EvaluationQuestions> = {
|
||||
readonly [ID in keyof Questions]: AnswerFor<Questions[ID]>
|
||||
}
|
||||
|
||||
export type EvaluationOptions = Record<string, unknown>
|
||||
|
||||
export interface EvaluationRoute<Options extends EvaluationOptions = EvaluationOptions> {
|
||||
readonly id: string
|
||||
readonly evaluate: (
|
||||
request: EvaluationRequestFor<Options>,
|
||||
execute: Execute,
|
||||
) => Effect.Effect<EvaluationResponse, AIError>
|
||||
}
|
||||
|
||||
export class EvaluationModel<Options extends EvaluationOptions = EvaluationOptions> {
|
||||
declare protected readonly _Options: (options: Options) => Options
|
||||
readonly id: ModelID
|
||||
readonly provider: ProviderID
|
||||
readonly route: EvaluationRoute<Options>
|
||||
readonly http?: HttpOptions
|
||||
|
||||
constructor(input: EvaluationModel.Input<Options>) {
|
||||
this.id = input.id
|
||||
this.provider = input.provider
|
||||
this.route = input.route
|
||||
this.http = input.http
|
||||
}
|
||||
|
||||
static make<Options extends EvaluationOptions = EvaluationOptions>(input: EvaluationModel.MakeInput<Options>) {
|
||||
return new EvaluationModel<Options>({
|
||||
id: ModelID.make(input.id),
|
||||
provider: ProviderID.make(input.provider),
|
||||
route: input.route,
|
||||
http: input.http,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export namespace EvaluationModel {
|
||||
export interface Input<Options extends EvaluationOptions = EvaluationOptions> {
|
||||
readonly id: ModelID
|
||||
readonly provider: ProviderID
|
||||
readonly route: EvaluationRoute<Options>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
|
||||
export interface MakeInput<Options extends EvaluationOptions = EvaluationOptions>
|
||||
extends Omit<Input<Options>, "id" | "provider"> {
|
||||
readonly id: string | ModelID
|
||||
readonly provider: string | ProviderID
|
||||
}
|
||||
}
|
||||
|
||||
export const EvaluationModelSchema = Schema.declare(
|
||||
(value): value is EvaluationModel => value instanceof EvaluationModel,
|
||||
{
|
||||
expected: "Evaluation.Model",
|
||||
},
|
||||
)
|
||||
|
||||
export class EvaluationRequest extends Schema.Class<EvaluationRequest>("Evaluation.Request")({
|
||||
model: EvaluationModelSchema,
|
||||
state: EvaluationInput,
|
||||
questions: EvaluationQuestions,
|
||||
options: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
http: Schema.optional(HttpOptions),
|
||||
}) {
|
||||
declare protected readonly _EvaluationRequest: void
|
||||
}
|
||||
|
||||
export type EvaluationModelOptions<Model> = Model extends EvaluationModel<infer Options> ? Options : never
|
||||
|
||||
export type EvaluationRequestFor<
|
||||
Options extends EvaluationOptions = EvaluationOptions,
|
||||
Questions extends EvaluationQuestions = EvaluationQuestions,
|
||||
> = Omit<EvaluationRequest, "model" | "questions" | "options"> & {
|
||||
readonly model: EvaluationModel<Options>
|
||||
readonly questions: Questions
|
||||
readonly options?: Options
|
||||
}
|
||||
|
||||
export type EvaluationRequestInput<
|
||||
Model extends object = EvaluationModel,
|
||||
Questions extends EvaluationQuestions = EvaluationQuestions,
|
||||
> = Omit<ConstructorParameters<typeof EvaluationRequest>[0], "model" | "questions" | "options" | "http"> & {
|
||||
readonly model: Model
|
||||
readonly questions: Questions
|
||||
readonly options?: NoInfer<EvaluationModelOptions<Model>>
|
||||
readonly http?: HttpOptions.Input
|
||||
} & (Model extends EvaluationModel<EvaluationModelOptions<Model>> ? unknown : never)
|
||||
|
||||
export class EvaluationRounding extends Schema.Class<EvaluationRounding>("Evaluation.Rounding")({
|
||||
probabilityDecimals: Schema.optional(Schema.Int),
|
||||
scoreDecimals: Schema.optional(Schema.Int),
|
||||
}) {}
|
||||
|
||||
export class EvaluationResponse extends Schema.Class<EvaluationResponse>("Evaluation.Response")({
|
||||
model: ModelID,
|
||||
answers: Schema.Record(Schema.String, EvaluationAnswer),
|
||||
usage: Schema.optional(Usage),
|
||||
rounding: Schema.optional(EvaluationRounding),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}) {}
|
||||
|
||||
export type EvaluationResponseFor<Questions extends EvaluationQuestions> = Omit<EvaluationResponse, "answers"> & {
|
||||
readonly answers: AnswersFor<Questions>
|
||||
}
|
||||
|
||||
export function request<const Model extends object, const Questions extends EvaluationQuestions>(
|
||||
input: EvaluationRequestInput<Model, Questions>,
|
||||
): EvaluationRequestFor<EvaluationModelOptions<Model>, Questions>
|
||||
export function request(input: EvaluationRequest): EvaluationRequest
|
||||
export function request(input: EvaluationRequest | EvaluationRequestInput) {
|
||||
if (input instanceof EvaluationRequest) return input
|
||||
return new EvaluationRequest({
|
||||
...input,
|
||||
model: input.model as unknown as EvaluationModel,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
}
|
||||
|
||||
export function run<const Model extends object, const Questions extends EvaluationQuestions>(
|
||||
input: EvaluationRequestInput<Model, Questions>,
|
||||
): Effect.Effect<EvaluationResponseFor<Questions>, AIError, Service>
|
||||
export function run(input: EvaluationRequest): Effect.Effect<EvaluationResponse, AIError, Service>
|
||||
export function run(input: EvaluationRequest | EvaluationRequestInput) {
|
||||
return Effect.try({
|
||||
try: () => (input instanceof EvaluationRequest ? input : request(input)),
|
||||
catch: (cause) =>
|
||||
new AIError({
|
||||
reason: new InvalidRequestError({
|
||||
message: cause instanceof Error ? cause.message : String(cause),
|
||||
cause,
|
||||
}),
|
||||
}),
|
||||
}).pipe(
|
||||
Effect.flatMap((request) =>
|
||||
EvaluationClient.evaluate(request as EvaluationRequestFor<EvaluationOptions, EvaluationQuestions>),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
export const Evaluation = {
|
||||
request,
|
||||
run,
|
||||
} as const
|
||||
@@ -0,0 +1,194 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import {
|
||||
ChoiceQuestion,
|
||||
EvaluationInput,
|
||||
EvaluationModel,
|
||||
EvaluationResponse,
|
||||
EvaluationRounding,
|
||||
ScoreQuestion,
|
||||
type EvaluationAnswer,
|
||||
type EvaluationOptions,
|
||||
} from "./evaluation.js"
|
||||
import { Auth, type Definition as AuthDefinition } from "../route/auth.js"
|
||||
import {
|
||||
AIError,
|
||||
HttpContext,
|
||||
HttpOptions,
|
||||
InvalidProviderOutputError,
|
||||
InvalidRequestError,
|
||||
ModelID,
|
||||
Usage,
|
||||
mergeJsonRecords,
|
||||
} from "../schema/index.js"
|
||||
|
||||
const Noul = Schema.Struct({
|
||||
type: Schema.Literal("noul"),
|
||||
instructions: EvaluationInput,
|
||||
criteria: Schema.optional(
|
||||
Schema.Struct({
|
||||
true: Schema.optional(Schema.NullOr(EvaluationInput)),
|
||||
false: Schema.optional(Schema.NullOr(EvaluationInput)),
|
||||
}),
|
||||
),
|
||||
})
|
||||
const Question = Schema.Union([
|
||||
ChoiceQuestion.pipe(
|
||||
Schema.refine((x): x is typeof x => Object.keys(x.criteria).length <= 255, {
|
||||
message: "System One Choice questions support at most 255 options",
|
||||
}),
|
||||
),
|
||||
ScoreQuestion.pipe(
|
||||
Schema.refine((x): x is typeof x => x.criteria.length <= 10, {
|
||||
message: "System One Score questions support at most 10 levels",
|
||||
}),
|
||||
),
|
||||
Noul,
|
||||
])
|
||||
const Request = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
model: Schema.String,
|
||||
state: EvaluationInput,
|
||||
questions: Schema.Record(Schema.String, Question),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Any)],
|
||||
)
|
||||
|
||||
const Probability = Schema.Number.check(Schema.isBetween({ minimum: 0, maximum: 1 }))
|
||||
const NoulAnswer = Schema.Struct({ type: Schema.Literal("noul"), noul: Probability })
|
||||
const Choice = Schema.Struct({
|
||||
type: Schema.Literal("choice"),
|
||||
choice: Schema.String,
|
||||
probabilities: Schema.Record(Schema.String, Probability),
|
||||
confidence: Schema.optional(Probability),
|
||||
})
|
||||
const Score = Schema.Struct({
|
||||
type: Schema.Literal("score"),
|
||||
score: Schema.Number,
|
||||
probabilities: Schema.Record(Schema.String, Probability),
|
||||
legend: Schema.optional(Schema.Record(Schema.String, Schema.Json)),
|
||||
confidence: Schema.optional(Probability),
|
||||
})
|
||||
const Answer = Schema.Union([NoulAnswer, Choice, Score]).pipe(Schema.toTaggedUnion("type"))
|
||||
const NativeUsage = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
input_tokens: Schema.optional(Schema.Number),
|
||||
output_tokens: Schema.optional(Schema.Number),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
)
|
||||
const Response = Schema.Struct({
|
||||
model: Schema.String,
|
||||
answers: Schema.Record(Schema.String, Answer),
|
||||
usage: Schema.optional(NativeUsage),
|
||||
id: Schema.optional(Schema.String),
|
||||
provider: Schema.optional(Schema.String),
|
||||
provider_metadata: Schema.optional(Schema.Record(Schema.String, Schema.Record(Schema.String, Schema.Unknown))),
|
||||
})
|
||||
|
||||
export interface ModelInput {
|
||||
readonly id: string | ModelID
|
||||
readonly provider: string
|
||||
readonly providerMetadataKey: string
|
||||
readonly auth: AuthDefinition
|
||||
readonly baseURL: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
|
||||
export const model = <Options extends EvaluationOptions = EvaluationOptions>(cfg: ModelInput) =>
|
||||
EvaluationModel.make<Options>({
|
||||
id: cfg.id,
|
||||
provider: cfg.provider,
|
||||
http: cfg.http,
|
||||
route: {
|
||||
id: "system-one",
|
||||
evaluate: (req, send) =>
|
||||
Effect.gen(function* () {
|
||||
const url = new URL(`${cfg.baseURL.replace(/\/$/, "")}/systemone`)
|
||||
Object.entries(req.http?.query ?? {}).forEach(([key, value]) => url.searchParams.set(key, value))
|
||||
const body = yield* Schema.encodeUnknownEffect(Schema.fromJsonString(Request))({
|
||||
...mergeJsonRecords(req.options, req.http?.body),
|
||||
model: req.model.id,
|
||||
state: req.state,
|
||||
questions: Object.fromEntries(
|
||||
Object.entries(req.questions).map(([id, x]) => [id, x.type === "boolean" ? { ...x, type: "noul" } : x]),
|
||||
),
|
||||
}).pipe(
|
||||
Effect.mapError(
|
||||
(cause) => new AIError({ reason: new InvalidRequestError({ message: cause.message, cause }) }),
|
||||
),
|
||||
)
|
||||
const headers = yield* Auth.toEffect(cfg.auth)({
|
||||
request: req,
|
||||
method: "POST",
|
||||
url: url.toString(),
|
||||
body,
|
||||
headers: Headers.fromInput({ ...cfg.headers, ...req.http?.headers }),
|
||||
})
|
||||
const res = yield* send(
|
||||
HttpClientRequest.post(url).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(body, "application/json"),
|
||||
),
|
||||
)
|
||||
const http = new HttpContext({ url: res.request.url, status: res.status, headers: res.headers })
|
||||
const fail = (message: string, cause: unknown, body?: string) =>
|
||||
new AIError({ reason: new InvalidProviderOutputError({ route: "system-one", message, body, http, cause }) })
|
||||
const text = yield* res.text.pipe(
|
||||
Effect.mapError((cause) => fail("Failed to read the System One response", cause)),
|
||||
)
|
||||
const data = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(Response))(text).pipe(
|
||||
Effect.mapError((cause) => fail("System One returned an invalid response", cause, text)),
|
||||
)
|
||||
|
||||
const confidence: Record<string, number> = {}
|
||||
const legend: Record<string, Record<string, Schema.Json>> = {}
|
||||
const answers = Object.fromEntries(
|
||||
Object.entries(data.answers).map(([id, answer]): [string, EvaluationAnswer] => {
|
||||
if (answer.type === "noul") return [id, { type: "boolean", probability: answer.noul }]
|
||||
if (answer.type === "choice") {
|
||||
if (answer.confidence !== undefined) confidence[id] = answer.confidence
|
||||
return [
|
||||
id,
|
||||
{
|
||||
type: "choice",
|
||||
choice: answer.choice,
|
||||
probabilities: answer.probabilities,
|
||||
},
|
||||
]
|
||||
}
|
||||
if (answer.confidence !== undefined) confidence[id] = answer.confidence
|
||||
if (answer.legend !== undefined) legend[id] = answer.legend
|
||||
return [id, { type: "score", score: answer.score, probabilities: answer.probabilities }]
|
||||
}),
|
||||
)
|
||||
const meta = {
|
||||
...(data.id === undefined ? {} : { responseId: data.id }),
|
||||
...(data.provider === undefined ? {} : { provider: data.provider }),
|
||||
...data.provider_metadata?.[cfg.providerMetadataKey],
|
||||
...(Object.keys(confidence).length === 0 ? {} : { confidence }),
|
||||
...(Object.keys(legend).length === 0 ? {} : { legend }),
|
||||
}
|
||||
return new EvaluationResponse({
|
||||
model: ModelID.make(data.model),
|
||||
answers,
|
||||
usage: data.usage
|
||||
? new Usage({
|
||||
inputTokens: data.usage.input_tokens,
|
||||
outputTokens: data.usage.output_tokens,
|
||||
totalTokens:
|
||||
data.usage.input_tokens === undefined && data.usage.output_tokens === undefined
|
||||
? undefined
|
||||
: (data.usage.input_tokens ?? 0) + (data.usage.output_tokens ?? 0),
|
||||
providerMetadata: { [cfg.providerMetadataKey]: data.usage },
|
||||
})
|
||||
: undefined,
|
||||
rounding: new EvaluationRounding({ probabilityDecimals: 2, scoreDecimals: 2 }),
|
||||
providerMetadata: Object.keys(meta).length === 0 ? undefined : { [cfg.providerMetadataKey]: meta },
|
||||
})
|
||||
}),
|
||||
},
|
||||
})
|
||||
|
||||
export const SystemOne = { model } as const
|
||||
@@ -0,0 +1,191 @@
|
||||
import { Clock, Duration, Effect, Schedule, Schema, Stream } from "effect"
|
||||
import { AIError, TimeoutError } from "./schema/errors.js"
|
||||
|
||||
export const Status = Schema.Literals(["queued", "running", "completed", "failed", "cancelled", "expired"])
|
||||
export type Status = Schema.Schema.Type<typeof Status>
|
||||
|
||||
/** Provider-neutral view of one generation observation. */
|
||||
export interface Snapshot {
|
||||
readonly id: string
|
||||
readonly status: Status
|
||||
/** Normalized 0..1 when the provider reports progress. */
|
||||
readonly progress?: number
|
||||
readonly position?: number
|
||||
readonly expiresAt?: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Route-owned generation operations for one generation. The media route decodes its serializable token once (from the
|
||||
* submission response or a `resume` input) and closes over it, so `Generation` never sees the token's shape.
|
||||
*/
|
||||
export interface Route<Response> {
|
||||
readonly status: Effect.Effect<Snapshot, AIError>
|
||||
readonly result: Effect.Effect<Response, AIError>
|
||||
readonly cancel?: Effect.Effect<void, AIError>
|
||||
/** Provider polling hint (e.g. `openai-poll-after-ms`) that overrides the default interval for the next poll. */
|
||||
readonly pollHint?: (snapshot: Snapshot) => Duration.Duration | undefined
|
||||
}
|
||||
|
||||
export interface Poll {
|
||||
readonly interval?: Duration.Input
|
||||
readonly timeout?: Duration.Input
|
||||
/** Full override of the polling schedule; `interval` and `pollHint` are ignored when supplied. */
|
||||
readonly schedule?: Schedule.Schedule<unknown, Snapshot>
|
||||
}
|
||||
|
||||
export interface AwaitOptions {
|
||||
readonly poll?: Poll
|
||||
}
|
||||
|
||||
export const DEFAULT_POLL_INTERVAL = Duration.seconds(5)
|
||||
export const DEFAULT_POLL_TIMEOUT = Duration.minutes(10)
|
||||
|
||||
export const QueuedEvent = Schema.Struct({
|
||||
type: Schema.tag("generation-queued"),
|
||||
id: Schema.String,
|
||||
position: Schema.optional(Schema.Number),
|
||||
}).annotate({ identifier: "Generation.Event.Queued" })
|
||||
|
||||
export const ProgressEvent = Schema.Struct({
|
||||
type: Schema.tag("generation-progress"),
|
||||
id: Schema.String,
|
||||
progress: Schema.optional(Schema.Number),
|
||||
}).annotate({ identifier: "Generation.Event.Progress" })
|
||||
|
||||
export type Observation = Schema.Schema.Type<typeof QueuedEvent> | Schema.Schema.Type<typeof ProgressEvent>
|
||||
|
||||
export type Event = Observation | { readonly type: "generation-finished"; readonly id: string; readonly status: Status }
|
||||
|
||||
const TERMINAL: ReadonlySet<Status> = new Set(["completed", "failed", "cancelled", "expired"])
|
||||
|
||||
export class Generation<Response> {
|
||||
readonly id: string
|
||||
readonly status: Status
|
||||
readonly progress?: number
|
||||
readonly position?: number
|
||||
readonly expiresAt?: number
|
||||
|
||||
constructor(
|
||||
readonly route: Route<Response>,
|
||||
/** Route-owned serializable JSON; pass it to the modality's `resume` from another process. */
|
||||
readonly token: unknown,
|
||||
snapshot: Snapshot,
|
||||
) {
|
||||
this.id = snapshot.id
|
||||
this.status = snapshot.status
|
||||
this.progress = snapshot.progress
|
||||
this.position = snapshot.position
|
||||
this.expiresAt = snapshot.expiresAt
|
||||
}
|
||||
|
||||
get snapshot(): Snapshot {
|
||||
return {
|
||||
id: this.id,
|
||||
status: this.status,
|
||||
progress: this.progress,
|
||||
position: this.position,
|
||||
expiresAt: this.expiresAt,
|
||||
}
|
||||
}
|
||||
|
||||
get terminal() {
|
||||
return TERMINAL.has(this.status)
|
||||
}
|
||||
|
||||
refresh(): Effect.Effect<Generation<Response>, AIError> {
|
||||
return this.route.status.pipe(Effect.map((snapshot) => new Generation(this.route, this.token, snapshot)))
|
||||
}
|
||||
|
||||
/** Fetch the result without polling; non-completed terminal generations fail with the provider's terminal body. */
|
||||
result(): Effect.Effect<Response, AIError> {
|
||||
return this.route.result
|
||||
}
|
||||
|
||||
/** Poll until the generation reaches a terminal status, then fetch the result. Fails with a `Timeout` reason on deadline. */
|
||||
await(options?: AwaitOptions): Effect.Effect<Response, AIError> {
|
||||
const timeout = Duration.fromInputUnsafe(options?.poll?.timeout ?? DEFAULT_POLL_TIMEOUT)
|
||||
const settled = this.terminal ? Effect.succeed(this) : this.poll(options?.poll)
|
||||
return settled.pipe(
|
||||
// Non-completed terminal states also go through `result` so the route can surface its provider failure body.
|
||||
Effect.flatMap((generation) => generation.result()),
|
||||
Effect.timeoutOrElse({ duration: timeout, orElse: () => this.timeoutError(timeout) }),
|
||||
)
|
||||
}
|
||||
|
||||
cancel(): Effect.Effect<void, AIError> {
|
||||
return this.route.cancel ?? Effect.void
|
||||
}
|
||||
|
||||
/**
|
||||
* Status observations as a stream, ending after the first terminal observation. Each poll is bounded by the time
|
||||
* remaining until `poll.timeout`, so a hung status request fails the stream instead of stalling it. (`Stream.interruptWhen`
|
||||
* would express this directly but deadlocks under `TestClock` when the source completes while the timer sleeps.)
|
||||
*/
|
||||
events(options?: AwaitOptions): Stream.Stream<Event, AIError> {
|
||||
if (this.terminal) return Stream.make(this.event())
|
||||
const timeout = Duration.fromInputUnsafe(options?.poll?.timeout ?? DEFAULT_POLL_TIMEOUT)
|
||||
return Stream.unwrap(
|
||||
Clock.currentTimeMillis.pipe(
|
||||
Effect.map((start) => {
|
||||
const deadline = start + Duration.toMillis(timeout)
|
||||
const refresh = Clock.currentTimeMillis.pipe(
|
||||
Effect.flatMap((now) =>
|
||||
this.refresh().pipe(
|
||||
Effect.timeoutOrElse({
|
||||
duration: Duration.millis(Math.max(0, deadline - now)),
|
||||
orElse: () => this.timeoutError(timeout),
|
||||
}),
|
||||
),
|
||||
),
|
||||
)
|
||||
return Stream.fromEffectSchedule(refresh, this.schedule(options?.poll)).pipe(
|
||||
Stream.takeUntil((generation) => generation.terminal),
|
||||
Stream.map((generation) => generation.event()),
|
||||
)
|
||||
}),
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
private event(): Event {
|
||||
if (this.terminal) return { type: "generation-finished", id: this.id, status: this.status }
|
||||
if (this.status === "queued") return { type: "generation-queued", id: this.id, position: this.position }
|
||||
return { type: "generation-progress", id: this.id, progress: this.progress }
|
||||
}
|
||||
|
||||
private timeoutError(timeout: Duration.Duration) {
|
||||
return new AIError({
|
||||
reason: new TimeoutError({
|
||||
message: `Generation ${this.id} did not finish within ${Duration.format(timeout)}`,
|
||||
timeoutMs: Duration.toMillis(timeout),
|
||||
}),
|
||||
})
|
||||
}
|
||||
|
||||
private poll(poll: Poll | undefined) {
|
||||
return this.refresh().pipe(
|
||||
Effect.repeat({ schedule: this.schedule(poll), until: (generation) => generation.terminal }),
|
||||
)
|
||||
}
|
||||
|
||||
private schedule(poll: Poll | undefined): Schedule.Schedule<unknown, Generation<Response>> {
|
||||
if (poll?.schedule) return poll.schedule.pipe(Schedule.setInputType<Generation<Response>>())
|
||||
const interval = poll?.interval ?? DEFAULT_POLL_INTERVAL
|
||||
const pollHint = this.route.pollHint
|
||||
const spaced = Schedule.spaced(interval).pipe(Schedule.setInputType<Generation<Response>>())
|
||||
if (!pollHint) return spaced
|
||||
return spaced.pipe(
|
||||
Schedule.modifyDelay((metadata) => Effect.succeed(pollHint(metadata.input.snapshot) ?? interval)),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
export const resultEvents = <Response, A>(
|
||||
generation: Generation<Response>,
|
||||
expand: (response: Response) => ReadonlyArray<A>,
|
||||
options?: AwaitOptions,
|
||||
): Stream.Stream<Observation | A, AIError> =>
|
||||
generation.events(options).pipe(
|
||||
Stream.filter((event): event is Observation => event.type !== "generation-finished"),
|
||||
Stream.concat(Stream.fromIterableEffect(Effect.map(generation.result(), expand))),
|
||||
)
|
||||
@@ -1,15 +1,21 @@
|
||||
import { Context, Effect, Layer } from "effect"
|
||||
import { Context, Effect, Layer, Stream } from "effect"
|
||||
import { RequestExecutor } from "./route/executor.js"
|
||||
import { mergeHttpOptions, type AIError } from "./schema/index.js"
|
||||
import { sanitizeSurrogates } from "./utils/sanitize.js"
|
||||
import type { ImageOptions, ImageRequest, ImageRequestFor, ImageResponse } from "./image.js"
|
||||
|
||||
export type Execute = RequestExecutor.Interface["execute"]
|
||||
import type { AIError } from "./schema/index.js"
|
||||
import {
|
||||
responseEvents,
|
||||
type ImageEvent,
|
||||
type ImageOptions,
|
||||
type ImageRequestFor,
|
||||
type ImageResponse,
|
||||
} from "./image.js"
|
||||
|
||||
export interface Interface {
|
||||
readonly generate: <Options extends ImageOptions>(
|
||||
request: ImageRequestFor<Options>,
|
||||
) => Effect.Effect<ImageResponse, AIError>
|
||||
readonly stream: <Options extends ImageOptions>(
|
||||
request: ImageRequestFor<Options>,
|
||||
) => Stream.Stream<ImageEvent, AIError>
|
||||
}
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/ImageClient") {}
|
||||
@@ -22,23 +28,27 @@ export const generate = <Options extends ImageOptions>(
|
||||
return yield* client.generate(request)
|
||||
})
|
||||
|
||||
export const stream = <Options extends ImageOptions>(
|
||||
request: ImageRequestFor<Options>,
|
||||
): Stream.Stream<ImageEvent, AIError, Service> =>
|
||||
Stream.unwrap(
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return client.stream(request)
|
||||
}),
|
||||
)
|
||||
|
||||
export const layer: Layer.Layer<Service, never, RequestExecutor.Service> = Layer.effect(
|
||||
Service,
|
||||
Effect.gen(function* () {
|
||||
const executor = yield* RequestExecutor.Service
|
||||
const generate = <Options extends ImageOptions>(request: ImageRequestFor<Options>) =>
|
||||
request.model.route.generate(request, executor.execute)
|
||||
return Service.of({
|
||||
generate: (request) =>
|
||||
request.model.route.generate(
|
||||
{
|
||||
...sanitizeSurrogates({
|
||||
...request,
|
||||
model: undefined,
|
||||
http: mergeHttpOptions(request.model.http, request.http),
|
||||
}),
|
||||
model: request.model,
|
||||
},
|
||||
executor.execute,
|
||||
),
|
||||
generate,
|
||||
// Inline routes have no partial frames yet; the stream is the completed response expanded into events.
|
||||
stream: (request) =>
|
||||
Stream.fromIterableEffect(Effect.map(generate(request), responseEvents)),
|
||||
})
|
||||
}),
|
||||
)
|
||||
@@ -47,4 +57,5 @@ export const ImageClient = {
|
||||
Service,
|
||||
layer,
|
||||
generate,
|
||||
stream,
|
||||
} as const
|
||||
|
||||
+119
-101
@@ -1,134 +1,115 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import {
|
||||
HttpOptions,
|
||||
InvalidRequestError,
|
||||
AIError,
|
||||
ModelID,
|
||||
ProviderID,
|
||||
ProviderMetadata,
|
||||
Usage,
|
||||
} from "./schema/index.js"
|
||||
import { ImageClient, Service, type Execute as ImageExecute } from "./image-client.js"
|
||||
import { Effect, Schema, Stream } from "effect"
|
||||
import { Media } from "./media.js"
|
||||
import { MediaModel, composeRoute, tryRequest } from "./media-model.js"
|
||||
import { MediaRoute } from "./route/media.js"
|
||||
import type { MediaProtocol } from "./route/media-protocol.js"
|
||||
import { AIError, HttpOptions, MediaUsage, ProviderMetadata } from "./schema/index.js"
|
||||
import { ImageClient, Service } from "./image-client.js"
|
||||
|
||||
export interface ImageRoute<Options extends ImageOptions = ImageOptions> {
|
||||
readonly id: string
|
||||
readonly generate: (request: ImageRequestFor<Options>, execute: ImageExecute) => Effect.Effect<ImageResponse, AIError>
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// Model
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type ImageOptions = Record<string, unknown>
|
||||
|
||||
export class ImageModel<Options extends ImageOptions = ImageOptions> {
|
||||
declare protected readonly _Options: (options: Options) => Options
|
||||
readonly id: ModelID
|
||||
readonly provider: ProviderID
|
||||
readonly route: ImageRoute<Options>
|
||||
readonly http?: HttpOptions
|
||||
export type ImageRoute<Options extends ImageOptions = ImageOptions> = MediaRoute.Route<
|
||||
ImageRequestFor<Options>,
|
||||
ImageResponse
|
||||
>
|
||||
|
||||
constructor(input: ImageModel.Input<Options>) {
|
||||
this.id = input.id
|
||||
this.provider = input.provider
|
||||
this.route = input.route
|
||||
this.http = input.http
|
||||
export class ImageModel<Options extends ImageOptions = ImageOptions> extends MediaModel<ImageRoute<Options>, Options> {
|
||||
declare protected readonly _ImageModel: void
|
||||
|
||||
static make<Options extends ImageOptions = ImageOptions>(input: MediaModel.Input<ImageRoute<Options>>) {
|
||||
return new ImageModel<Options>(input)
|
||||
}
|
||||
|
||||
static make<Options extends ImageOptions = ImageOptions>(input: ImageModel.MakeInput<Options>) {
|
||||
/** Compose an inline image protocol with its canonical path into a model for one deployment. */
|
||||
static fromRoute<Options extends ImageOptions = ImageOptions>(
|
||||
route: ImageModel.RouteInput<Options>,
|
||||
input: MediaRoute.ModelInput,
|
||||
) {
|
||||
return new ImageModel<Options>({
|
||||
id: ModelID.make(input.id),
|
||||
provider: ProviderID.make(input.provider),
|
||||
route: input.route,
|
||||
id: input.id,
|
||||
provider: route.provider,
|
||||
http: input.http,
|
||||
route: composeRoute(MediaRoute.inline, route, input),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export namespace ImageModel {
|
||||
export interface Input<Options extends ImageOptions = ImageOptions> {
|
||||
readonly id: ModelID
|
||||
readonly provider: ProviderID
|
||||
readonly route: ImageRoute<Options>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
|
||||
export interface MakeInput<Options extends ImageOptions = ImageOptions>
|
||||
extends Omit<Input<Options>, "id" | "provider"> {
|
||||
readonly id: string | ModelID
|
||||
readonly provider: string | ProviderID
|
||||
}
|
||||
export type RouteInput<Options extends ImageOptions = ImageOptions> = MediaModel.RouteInput<
|
||||
ImageRequestFor<Options>,
|
||||
MediaProtocol.Inline<ImageRequestFor<Options>, ImageResponse>
|
||||
>
|
||||
}
|
||||
|
||||
export const ImageModelSchema = Schema.declare((value): value is ImageModel => value instanceof ImageModel, {
|
||||
expected: "Image.Model",
|
||||
})
|
||||
|
||||
const ImageBytesInput = Schema.Struct({
|
||||
type: Schema.Literal("bytes"),
|
||||
data: Schema.Uint8Array,
|
||||
mediaType: Schema.String,
|
||||
})
|
||||
const ImageUrlInput = Schema.Struct({
|
||||
type: Schema.Literal("url"),
|
||||
url: Schema.String,
|
||||
})
|
||||
const ImageFileIDInput = Schema.Struct({
|
||||
type: Schema.Literal("file-id"),
|
||||
id: Schema.String,
|
||||
})
|
||||
const ImageFileURIInput = Schema.Struct({
|
||||
type: Schema.Literal("file-uri"),
|
||||
uri: Schema.String,
|
||||
mediaType: Schema.String,
|
||||
})
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const ImageInputSchema = Schema.Union([
|
||||
ImageBytesInput,
|
||||
ImageUrlInput,
|
||||
ImageFileIDInput,
|
||||
ImageFileURIInput,
|
||||
]).pipe(Schema.toTaggedUnion("type"))
|
||||
export type ImageInput = Schema.Schema.Type<typeof ImageInputSchema>
|
||||
export type ImageSize = `${number}x${number}`
|
||||
export const ImageSize = Schema.declare<ImageSize>(
|
||||
(value): value is ImageSize => typeof value === "string" && /^\d+x\d+$/.test(value),
|
||||
{ title: "ImageSize" },
|
||||
)
|
||||
|
||||
export const ImageInput = {
|
||||
bytes: (data: Uint8Array, mediaType: string): ImageInput => ({ type: "bytes", data, mediaType }),
|
||||
url: (url: string): ImageInput => ({ type: "url", url }),
|
||||
file: (id: string): ImageInput => ({ type: "file-id", id }),
|
||||
fileUri: (uri: string, mediaType: string): ImageInput => ({ type: "file-uri", uri, mediaType }),
|
||||
} as const
|
||||
export type ImageAspectRatio = Media.AspectRatio
|
||||
export const ImageAspectRatio = Media.AspectRatio
|
||||
|
||||
export type ImageFormat = "png" | "jpeg" | "webp" | (string & {})
|
||||
|
||||
export class ImageRequest extends Schema.Class<ImageRequest>("Image.Request")({
|
||||
model: ImageModelSchema,
|
||||
prompt: Schema.String,
|
||||
images: Schema.optional(Schema.Array(ImageInputSchema)),
|
||||
options: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
/** Edit sources or style/subject references, in order. */
|
||||
images: Schema.optional(Schema.Array(Media.AssetSchema)),
|
||||
/** Inpainting mask; routes that cannot honor it fail with `UnsupportedOperation`. */
|
||||
mask: Schema.optional(Media.AssetSchema),
|
||||
n: Schema.optional(Schema.Int),
|
||||
size: Schema.optional(ImageSize),
|
||||
aspectRatio: Schema.optional(ImageAspectRatio),
|
||||
seed: Schema.optional(Schema.Number),
|
||||
format: Schema.optional(Schema.String),
|
||||
providerOptions: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
http: Schema.optional(HttpOptions),
|
||||
}) {
|
||||
declare protected readonly _ImageRequest: void
|
||||
}
|
||||
|
||||
export type ImageRequestFor<Options extends ImageOptions = ImageOptions> = Omit<ImageRequest, "model" | "options"> & {
|
||||
export type ImageRequestFor<Options extends ImageOptions = ImageOptions> = Omit<
|
||||
ImageRequest,
|
||||
"model" | "providerOptions"
|
||||
> & {
|
||||
readonly model: ImageModel<Options>
|
||||
readonly options?: Options
|
||||
readonly providerOptions?: Options
|
||||
}
|
||||
|
||||
export type ImageModelOptions<Model> = Model extends ImageModel<infer Options> ? Options : never
|
||||
|
||||
export type ImageRequestInput<Model extends object = ImageModel> = Omit<
|
||||
export type ImageRequestInput<Model extends ImageModel = ImageModel> = Omit<
|
||||
ConstructorParameters<typeof ImageRequest>[0],
|
||||
"model" | "options" | "http"
|
||||
"model" | "providerOptions" | "http"
|
||||
> & {
|
||||
readonly model: Model
|
||||
readonly options?: NoInfer<ImageModelOptions<Model>>
|
||||
readonly format?: ImageFormat
|
||||
readonly providerOptions?: NoInfer<ImageModelOptions<Model>>
|
||||
readonly http?: HttpOptions.Input
|
||||
} & (Model extends ImageModel<ImageModelOptions<Model>> ? unknown : never)
|
||||
}
|
||||
|
||||
export class GeneratedImage extends Schema.Class<GeneratedImage>("Image.Generated")({
|
||||
mediaType: Schema.String,
|
||||
data: Schema.Union([Schema.String, Schema.Uint8Array]),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}) {}
|
||||
// ---------------------------------------------------------------------------
|
||||
// Response and events
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export class ImageResponse extends Schema.Class<ImageResponse>("Image.Response")({
|
||||
images: Schema.Array(GeneratedImage),
|
||||
usage: Schema.optional(Usage),
|
||||
images: Schema.Array(Media.AssetSchema),
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}) {
|
||||
get image() {
|
||||
@@ -136,7 +117,43 @@ export class ImageResponse extends Schema.Class<ImageResponse>("Image.Response")
|
||||
}
|
||||
}
|
||||
|
||||
export function request<const Model extends object>(
|
||||
export const ImageOutputEvent = Schema.Struct({
|
||||
type: Schema.tag("image"),
|
||||
index: Schema.Number,
|
||||
image: Media.AssetSchema,
|
||||
}).annotate({ identifier: "Image.Event.Image" })
|
||||
|
||||
export const ImageFinishEvent = Schema.Struct({
|
||||
type: Schema.tag("finish"),
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}).annotate({ identifier: "Image.Event.Finish" })
|
||||
|
||||
const imageEventTagged = Schema.Union([ImageOutputEvent, ImageFinishEvent]).pipe(Schema.toTaggedUnion("type"))
|
||||
export const ImageEvent = Object.assign(imageEventTagged, {
|
||||
is: {
|
||||
image: imageEventTagged.guards.image,
|
||||
finish: imageEventTagged.guards.finish,
|
||||
},
|
||||
})
|
||||
export type ImageEvent = Schema.Schema.Type<typeof imageEventTagged>
|
||||
|
||||
/** Inline routes produce every image at once; expand the response into the streaming event shape. */
|
||||
export const responseEvents = (response: ImageResponse): ReadonlyArray<ImageEvent> => [
|
||||
...response.images.map((image, index) => ImageOutputEvent.make({ index, image })),
|
||||
ImageFinishEvent.make({
|
||||
usage: response.usage,
|
||||
notices: response.notices,
|
||||
providerMetadata: response.providerMetadata,
|
||||
}),
|
||||
]
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request-shaped call API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export function request<const Model extends ImageModel>(
|
||||
input: ImageRequestInput<Model>,
|
||||
): ImageRequestFor<ImageModelOptions<Model>>
|
||||
export function request(input: ImageRequest): ImageRequest
|
||||
@@ -144,29 +161,30 @@ export function request(input: ImageRequest | ImageRequestInput) {
|
||||
if (input instanceof ImageRequest) return input
|
||||
return new ImageRequest({
|
||||
...input,
|
||||
model: input.model as unknown as ImageModel,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
}
|
||||
|
||||
export function generate<const Model extends object>(
|
||||
const requestEffect = (input: ImageRequest | ImageRequestInput) => tryRequest(() => request(input))
|
||||
|
||||
export function generate<const Model extends ImageModel>(
|
||||
input: ImageRequestInput<Model>,
|
||||
): Effect.Effect<ImageResponse, AIError, Service>
|
||||
export function generate(input: ImageRequest): Effect.Effect<ImageResponse, AIError, Service>
|
||||
export function generate(input: ImageRequest | ImageRequestInput) {
|
||||
return Effect.try({
|
||||
try: () => (input instanceof ImageRequest ? input : request(input)),
|
||||
catch: (error) =>
|
||||
new AIError({
|
||||
reason: new InvalidRequestError({
|
||||
message: error instanceof Error ? error.message : String(error),
|
||||
cause: error,
|
||||
}),
|
||||
}),
|
||||
}).pipe(Effect.flatMap((request) => ImageClient.generate(request as unknown as ImageRequestFor<ImageOptions>)))
|
||||
return requestEffect(input).pipe(Effect.flatMap((request) => ImageClient.generate(request)))
|
||||
}
|
||||
|
||||
export function stream<const Model extends ImageModel>(
|
||||
input: ImageRequestInput<Model>,
|
||||
): Stream.Stream<ImageEvent, AIError, Service>
|
||||
export function stream(input: ImageRequest): Stream.Stream<ImageEvent, AIError, Service>
|
||||
export function stream(input: ImageRequest | ImageRequestInput) {
|
||||
return Stream.unwrap(requestEffect(input).pipe(Effect.map((request) => ImageClient.stream(request))))
|
||||
}
|
||||
|
||||
export const Image = {
|
||||
request,
|
||||
generate,
|
||||
stream,
|
||||
} as const
|
||||
|
||||
@@ -11,9 +11,91 @@ export type {
|
||||
Service as LLMClientService,
|
||||
} from "./route/client.js"
|
||||
export * from "./schema/index.js"
|
||||
export { GeneratedImage, ImageInput, ImageInputSchema, ImageModel, ImageRequest, ImageResponse } from "./image.js"
|
||||
export type { ImageModelOptions, ImageOptions, ImageRequestFor, ImageRequestInput, ImageRoute } from "./image.js"
|
||||
export {
|
||||
ImageAspectRatio,
|
||||
ImageEvent,
|
||||
ImageModel,
|
||||
ImageModelSchema,
|
||||
ImageRequest,
|
||||
ImageResponse,
|
||||
ImageSize,
|
||||
} from "./image.js"
|
||||
export type {
|
||||
ImageFormat,
|
||||
ImageModelOptions,
|
||||
ImageOptions,
|
||||
ImageRequestFor,
|
||||
ImageRequestInput,
|
||||
ImageRoute,
|
||||
} from "./image.js"
|
||||
export { Image } from "./image.js"
|
||||
export { VideoClient } from "./video-client.js"
|
||||
export {
|
||||
VideoAspectRatio,
|
||||
VideoEvent,
|
||||
VideoFrames,
|
||||
VideoModel,
|
||||
VideoModelSchema,
|
||||
VideoRequest,
|
||||
VideoResponse,
|
||||
} from "./video.js"
|
||||
export type {
|
||||
VideoModelOptions,
|
||||
VideoOptions,
|
||||
VideoRequestFor,
|
||||
VideoRequestInput,
|
||||
VideoResolution,
|
||||
VideoRoute,
|
||||
} from "./video.js"
|
||||
export { Video } from "./video.js"
|
||||
export { SpeechClient } from "./speech-client.js"
|
||||
export {
|
||||
SpeechEvent,
|
||||
SpeechModel,
|
||||
SpeechModelSchema,
|
||||
SpeechRequest,
|
||||
SpeechResponse,
|
||||
SpeechTimestamp,
|
||||
SpeechVoice,
|
||||
} from "./speech.js"
|
||||
export type {
|
||||
SpeechFormat,
|
||||
SpeechModelOptions,
|
||||
SpeechOptions,
|
||||
SpeechRequestFor,
|
||||
SpeechRequestInput,
|
||||
SpeechRoute,
|
||||
} from "./speech.js"
|
||||
export { Speech } from "./speech.js"
|
||||
export { TranscriptionClient } from "./transcription-client.js"
|
||||
export {
|
||||
TranscriptionEvent,
|
||||
TranscriptionModel,
|
||||
TranscriptionModelSchema,
|
||||
TranscriptionRequest,
|
||||
TranscriptionResponse,
|
||||
TranscriptionSegment,
|
||||
TranscriptionTimestamps,
|
||||
TranscriptionWord,
|
||||
} from "./transcription.js"
|
||||
export type {
|
||||
TranscriptionModelOptions,
|
||||
TranscriptionOptions,
|
||||
TranscriptionRequestFor,
|
||||
TranscriptionRequestInput,
|
||||
TranscriptionRoute,
|
||||
} from "./transcription.js"
|
||||
export { Transcription } from "./transcription.js"
|
||||
export { Media } from "./media.js"
|
||||
export { Generation } from "./generation.js"
|
||||
export type {
|
||||
AwaitOptions as GenerationAwaitOptions,
|
||||
Event as GenerationEvent,
|
||||
Poll,
|
||||
Route as GenerationRoute,
|
||||
Snapshot as GenerationSnapshot,
|
||||
Status as GenerationStatus,
|
||||
} from "./generation.js"
|
||||
export { Tool, ToolFailure, toDefinitions } from "./tool.js"
|
||||
export { ToolRuntime } from "./tool-runtime.js"
|
||||
export type { DispatchResult as ToolDispatchResult, ToolSettlement } from "./tool-runtime.js"
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
import { Effect } from "effect"
|
||||
import { Endpoint } from "./route/endpoint.js"
|
||||
import type { MediaRoute } from "./route/media.js"
|
||||
import type { MediaProtocol } from "./route/media-protocol.js"
|
||||
import { AIError, HttpOptions, InvalidRequestError, ModelID, ProviderID } from "./schema/index.js"
|
||||
|
||||
/**
|
||||
* What every media model carries: ids, the configured route, and deployment `http` overlays. Modality classes
|
||||
* (`ImageModel`, `VideoModel`, `SpeechModel`) extend it with their route type and a nominal marker so one cannot stand
|
||||
* in for the other in requests.
|
||||
*/
|
||||
export class MediaModel<Route, Options> {
|
||||
declare protected readonly _Options: (options: Options) => Options
|
||||
readonly id: ModelID
|
||||
readonly provider: ProviderID
|
||||
readonly route: Route
|
||||
readonly http?: HttpOptions
|
||||
|
||||
constructor(input: MediaModel.Input<Route>) {
|
||||
this.id = ModelID.make(input.id)
|
||||
this.provider = ProviderID.make(input.provider)
|
||||
this.route = input.route
|
||||
this.http = input.http
|
||||
}
|
||||
}
|
||||
|
||||
export namespace MediaModel {
|
||||
export interface Input<Route> {
|
||||
readonly id: string | ModelID
|
||||
readonly provider: string | ProviderID
|
||||
readonly route: Route
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
|
||||
/** A protocol plus its canonical start path; `ModelInput.baseURL` overrides `baseURL` per deployment. */
|
||||
export interface RouteInput<Request extends MediaRoute.MediaRequest, Protocol> {
|
||||
readonly id: string
|
||||
readonly provider: string | ProviderID
|
||||
readonly protocol: Protocol
|
||||
readonly path: Endpoint.EndpointPart<MediaProtocol.Body, Request>
|
||||
readonly baseURL?: string
|
||||
/** Headers the protocol requires on every call, such as a pinned API version; deployment headers win. */
|
||||
readonly headers?: Record<string, string>
|
||||
}
|
||||
}
|
||||
|
||||
/** Compose a protocol route input with one deployment through `MediaRoute.inline`, `queued`, or `stream`. */
|
||||
export const composeRoute = <Request extends MediaRoute.MediaRequest, Protocol, Route>(
|
||||
compose: (input: MediaRoute.Composition<Request> & { readonly protocol: Protocol }) => Route,
|
||||
route: MediaModel.RouteInput<Request, Protocol>,
|
||||
input: MediaRoute.ModelInput,
|
||||
): Route =>
|
||||
compose({
|
||||
id: route.id,
|
||||
provider: route.provider,
|
||||
protocol: route.protocol,
|
||||
endpoint: Endpoint.path(route.path, { baseURL: input.baseURL ?? route.baseURL }),
|
||||
auth: input.auth,
|
||||
headers:
|
||||
route.headers === undefined && input.headers === undefined ? undefined : { ...route.headers, ...input.headers },
|
||||
})
|
||||
|
||||
/** Lift a synchronous Schema-class constructor into a typed `InvalidRequest` failure. */
|
||||
export const tryRequest = <A>(make: () => A): Effect.Effect<A, AIError> =>
|
||||
Effect.try({
|
||||
try: make,
|
||||
catch: (error) =>
|
||||
new AIError({
|
||||
reason: new InvalidRequestError({
|
||||
message: error instanceof Error ? error.message : String(error),
|
||||
cause: error,
|
||||
}),
|
||||
}),
|
||||
})
|
||||
@@ -0,0 +1,323 @@
|
||||
export * as Media from "./media.js"
|
||||
|
||||
import { Effect, Encoding, FileSystem, Schema, SchemaGetter } from "effect"
|
||||
import { HttpClientRequest } from "effect/unstable/http"
|
||||
import { ProviderID } from "./schema/ids.js"
|
||||
import { AIError, HttpContext, InvalidProviderOutputError, InvalidRequestError } from "./schema/errors.js"
|
||||
import { ProviderMetadata } from "./schema/options.js"
|
||||
import { Service } from "./route/executor-service.js"
|
||||
import { detectMediaType, extensionMediaType } from "./utils/media-type.js"
|
||||
|
||||
export { detectMediaType } from "./utils/media-type.js"
|
||||
|
||||
const OCTET_STREAM = "application/octet-stream"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Source — the serializable wire/persistence form of a media asset
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const BytesSource = Schema.Struct({
|
||||
type: Schema.Literal("bytes"),
|
||||
data: Schema.Uint8Array,
|
||||
mediaType: Schema.String,
|
||||
})
|
||||
|
||||
const Base64Source = Schema.Struct({
|
||||
type: Schema.Literal("base64"),
|
||||
data: Schema.String,
|
||||
mediaType: Schema.String,
|
||||
})
|
||||
|
||||
const UrlSource = Schema.Struct({
|
||||
type: Schema.Literal("url"),
|
||||
url: Schema.String,
|
||||
mediaType: Schema.optional(Schema.String),
|
||||
/** Epoch milliseconds after which the provider no longer serves the URL. */
|
||||
expiresAt: Schema.optional(Schema.Number),
|
||||
})
|
||||
|
||||
/** A provider-side handle: OpenAI `file_id`, Gemini file URI, `gs://`, `runway://`, or a prior generation id. */
|
||||
const RefSource = Schema.Struct({
|
||||
type: Schema.Literal("ref"),
|
||||
provider: ProviderID,
|
||||
id: Schema.String,
|
||||
mediaType: Schema.optional(Schema.String),
|
||||
})
|
||||
|
||||
export const Source = Schema.Union([BytesSource, Base64Source, UrlSource, RefSource])
|
||||
.pipe(Schema.toTaggedUnion("type"))
|
||||
.annotate({ identifier: "Media.Source" })
|
||||
export type Source = Schema.Schema.Type<typeof Source>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Kind, Info, Notice
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type AspectRatio = `${number}:${number}`
|
||||
export const AspectRatio = Schema.declare<AspectRatio>(
|
||||
(value): value is AspectRatio => typeof value === "string" && /^\d+(?:\.\d+)?:\d+(?:\.\d+)?$/.test(value),
|
||||
{ title: "Media.AspectRatio" },
|
||||
)
|
||||
|
||||
export const Kind = Schema.Literals(["image", "video", "audio", "document", "other"])
|
||||
export type Kind = Schema.Schema.Type<typeof Kind>
|
||||
|
||||
export const kindOf = (mediaType: string): Kind => {
|
||||
const lower = mediaType.toLowerCase()
|
||||
if (lower.startsWith("image/")) return "image"
|
||||
if (lower.startsWith("video/")) return "video"
|
||||
if (lower.startsWith("audio/")) return "audio"
|
||||
if (lower === "application/pdf" || lower.startsWith("text/")) return "document"
|
||||
return "other"
|
||||
}
|
||||
|
||||
/** Container-independent facts about the payload; raw PCM audio relies on these because it has no header. */
|
||||
export const Info = Schema.Struct({
|
||||
width: Schema.optional(Schema.Number),
|
||||
height: Schema.optional(Schema.Number),
|
||||
durationSeconds: Schema.optional(Schema.Number),
|
||||
sampleRate: Schema.optional(Schema.Number),
|
||||
channels: Schema.optional(Schema.Number),
|
||||
encoding: Schema.optional(Schema.String),
|
||||
format: Schema.optional(Schema.String),
|
||||
}).annotate({ identifier: "Media.Info" })
|
||||
export type Info = Schema.Schema.Type<typeof Info>
|
||||
|
||||
/** A provider-side partial result such as stripped audio or a moderated sample; never a silent drop. */
|
||||
export const Notice = Schema.Struct({
|
||||
type: Schema.Literals(["moderated", "filtered", "other"]),
|
||||
message: Schema.String,
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}).annotate({ identifier: "Media.Notice" })
|
||||
export type Notice = Schema.Schema.Type<typeof Notice>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Asset
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const invalid = (message: string, cause?: unknown) =>
|
||||
new AIError({ reason: new InvalidRequestError({ message, cause }) })
|
||||
|
||||
/** Synchronous view of an inline payload; `undefined` for `url` and `ref` sources, which carry no local bytes. */
|
||||
export interface Inline {
|
||||
readonly mime: string
|
||||
readonly base64: string
|
||||
readonly dataUrl: string
|
||||
}
|
||||
|
||||
export class Asset {
|
||||
readonly source: Source
|
||||
/** Derived from the source: declared type, sniffed magic bytes, then `application/octet-stream`. */
|
||||
readonly mediaType: string
|
||||
readonly kind: Kind
|
||||
readonly info?: Info
|
||||
/** Epoch milliseconds after which a `url` source stops resolving. */
|
||||
readonly expiresAt?: number
|
||||
readonly providerMetadata?: ProviderMetadata
|
||||
/** Transient download credentials for `url` sources; see `Asset.Input.headers`. */
|
||||
readonly headers?: Record<string, string>
|
||||
|
||||
// Derived payload forms are cached on the instance because every protocol lowering re-reads the same payload. The
|
||||
// cache is check-then-set (concurrent first reads of a `url` source may both download) and is never observable
|
||||
// through `source`, so round-tripping through `Media.from(asset.source)` stays lossless.
|
||||
#bytes: Uint8Array | undefined
|
||||
#base64: string | undefined
|
||||
|
||||
constructor(input: Asset.Input) {
|
||||
this.source = input.source
|
||||
this.mediaType =
|
||||
input.source.mediaType ??
|
||||
(input.source.type === "bytes" ? detectMediaType(input.source.data) : undefined) ??
|
||||
OCTET_STREAM
|
||||
this.kind = kindOf(this.mediaType)
|
||||
this.info = input.info
|
||||
this.expiresAt = input.source.type === "url" ? input.source.expiresAt : undefined
|
||||
this.providerMetadata = input.providerMetadata
|
||||
this.headers = input.source.type === "url" ? input.headers : undefined
|
||||
}
|
||||
|
||||
/** Inline payload without effects, for protocols that embed base64 or data URLs directly. */
|
||||
inline(): Inline | undefined {
|
||||
const source = this.source
|
||||
if (source.type !== "bytes" && source.type !== "base64") return undefined
|
||||
const base64 = source.type === "base64" ? source.data : (this.#base64 ??= Encoding.encodeBase64(source.data))
|
||||
const mime = this.mediaType.toLowerCase()
|
||||
return { mime, base64, dataUrl: `data:${mime};base64,${base64}` }
|
||||
}
|
||||
|
||||
/** Decoded payload; downloads `url` sources through the request executor and caches the result. */
|
||||
bytes(): Effect.Effect<Uint8Array, AIError, Service> {
|
||||
return Effect.suspend(() => {
|
||||
const source = this.source
|
||||
if (source.type === "bytes") return Effect.succeed(source.data)
|
||||
if (this.#bytes !== undefined) return Effect.succeed(this.#bytes)
|
||||
if (source.type === "ref")
|
||||
return Effect.fail(invalid(`Cannot materialize provider ref ${source.provider}:${source.id}`))
|
||||
const decoded =
|
||||
source.type === "base64"
|
||||
? Effect.fromResult(Encoding.decodeBase64(source.data)).pipe(
|
||||
Effect.mapError((cause) => invalid(`Media asset contains invalid base64 data`, cause)),
|
||||
)
|
||||
: download(source, this.headers)
|
||||
return decoded.pipe(Effect.tap((data) => Effect.sync(() => (this.#bytes = data))))
|
||||
})
|
||||
}
|
||||
|
||||
base64(): Effect.Effect<string, AIError, Service> {
|
||||
return Effect.suspend(() => {
|
||||
const source = this.source
|
||||
if (source.type === "base64") return Effect.succeed(source.data)
|
||||
if (this.#base64 !== undefined) return Effect.succeed(this.#base64)
|
||||
return this.bytes().pipe(Effect.map((data) => (this.#base64 = Encoding.encodeBase64(data))))
|
||||
})
|
||||
}
|
||||
|
||||
dataUrl(): Effect.Effect<string, AIError, Service> {
|
||||
return this.base64().pipe(Effect.map((data) => `data:${this.mediaType};base64,${data}`))
|
||||
}
|
||||
|
||||
/**
|
||||
* The `AssetEncoded` JSON form with `bytes` sources as base64, matching `Schema.toCodecJson(AssetSchema)`, so a
|
||||
* plain `JSON.stringify` of messages or events stays lossless and decodes back through the JSON codec.
|
||||
*/
|
||||
toJSON() {
|
||||
const source = this.source
|
||||
return {
|
||||
source: source.type === "bytes" ? { ...source, data: Encoding.encodeBase64(source.data) } : source,
|
||||
info: this.info,
|
||||
providerMetadata: this.providerMetadata,
|
||||
}
|
||||
}
|
||||
|
||||
/** Pull `url` sources into owned bytes before the URL expires. Inline sources return themselves. */
|
||||
materialize(): Effect.Effect<Asset, AIError, Service> {
|
||||
if (this.source.type === "bytes" || this.source.type === "base64") return Effect.succeed(this)
|
||||
return this.bytes().pipe(
|
||||
Effect.map((data) =>
|
||||
bytes(data, this.source.mediaType, { info: this.info, providerMetadata: this.providerMetadata }),
|
||||
),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
export namespace Asset {
|
||||
export interface Input {
|
||||
readonly source: Source
|
||||
readonly info?: Info
|
||||
readonly providerMetadata?: ProviderMetadata
|
||||
/**
|
||||
* Headers required to download a `url` source, such as the provider API key Veo demands for its file URIs.
|
||||
* They are runtime-only: never part of `source`, `toJSON()`, or `AssetSchema`, so a persisted asset cannot leak
|
||||
* credentials and cannot be downloaded again after a round-trip. Call `materialize()` before persisting.
|
||||
*/
|
||||
readonly headers?: Record<string, string>
|
||||
}
|
||||
}
|
||||
|
||||
/** JSON form of an asset: the serializable `Source` plus caller-supplied metadata. `bytes` sources encode as base64. */
|
||||
export const AssetEncoded = Schema.Struct({
|
||||
source: Source,
|
||||
info: Schema.optional(Info),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}).annotate({ identifier: "Media.AssetEncoded" })
|
||||
|
||||
const encodeAsset = (asset: Asset): typeof AssetEncoded.Type => ({
|
||||
source: asset.source,
|
||||
info: asset.info,
|
||||
providerMetadata: asset.providerMetadata,
|
||||
})
|
||||
|
||||
const AssetInstance = Schema.declare((value): value is Asset => value instanceof Asset, {
|
||||
expected: "Media.Asset",
|
||||
})
|
||||
|
||||
/** `Asset` in the type domain and `AssetEncoded` on the wire, so messages and events holding assets serialize. */
|
||||
export const AssetSchema = AssetEncoded.pipe(
|
||||
Schema.decodeTo(AssetInstance, {
|
||||
decode: SchemaGetter.transform((encoded) => new Asset(encoded)),
|
||||
encode: SchemaGetter.transform(encodeAsset),
|
||||
}),
|
||||
)
|
||||
|
||||
const download = Effect.fn("Media.download")(function* (
|
||||
source: Extract<Source, { readonly type: "url" }>,
|
||||
headers: Record<string, string> | undefined,
|
||||
) {
|
||||
const executor = yield* Service
|
||||
const response = yield* executor.execute(
|
||||
HttpClientRequest.get(source.url).pipe(HttpClientRequest.setHeaders(headers ?? {})),
|
||||
)
|
||||
const buffer = yield* response.arrayBuffer.pipe(
|
||||
Effect.mapError(
|
||||
(cause) =>
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({
|
||||
message: `Failed to read media from ${source.url}`,
|
||||
http: new HttpContext({ url: response.request.url, status: response.status, headers: response.headers }),
|
||||
cause,
|
||||
}),
|
||||
}),
|
||||
),
|
||||
)
|
||||
return new Uint8Array(buffer)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Constructors
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type AssetOptions = Omit<Asset.Input, "source">
|
||||
|
||||
export const from = (source: Source, options?: AssetOptions) => new Asset({ ...options, source })
|
||||
|
||||
export const bytes = (data: Uint8Array, mediaType?: string, options?: AssetOptions) =>
|
||||
from({ type: "bytes", data, mediaType: mediaType ?? detectMediaType(data) ?? OCTET_STREAM }, options)
|
||||
|
||||
export const base64 = (data: string, mediaType: string, options?: AssetOptions) =>
|
||||
from({ type: "base64", data, mediaType }, options)
|
||||
|
||||
export const url = (
|
||||
value: string,
|
||||
options?: AssetOptions & Omit<Extract<Source, { readonly type: "url" }>, "type" | "url">,
|
||||
) => {
|
||||
const { mediaType, expiresAt, ...rest } = options ?? {}
|
||||
return from({ type: "url", url: value, mediaType, expiresAt }, rest)
|
||||
}
|
||||
|
||||
export const ref = (provider: string | ProviderID, id: string, mediaType?: string, options?: AssetOptions) =>
|
||||
from({ type: "ref", provider: ProviderID.make(provider), id, mediaType }, options)
|
||||
|
||||
const DATA_URL = /^data:([^;,]+)(?:;[^,]*)*;base64,(.*)$/s
|
||||
|
||||
/** Parse a `data:<mime>;base64,<data>` URL, or `undefined` when the value is not a base64 data URL. */
|
||||
export const parseDataUrl = (value: string, options?: AssetOptions) => {
|
||||
const match = DATA_URL.exec(value)
|
||||
return match === null ? undefined : base64(match[2], match[1], options)
|
||||
}
|
||||
|
||||
/** Parse a `data:<mime>;base64,<data>` URL. Malformed input throws a typed `AIError` because constructors are sync. */
|
||||
export const fromDataUrl = (dataUrl: string, options?: AssetOptions) => {
|
||||
const asset = parseDataUrl(dataUrl, options)
|
||||
if (asset === undefined) throw invalid("Media data URLs must contain a MIME type and base64 data")
|
||||
return asset
|
||||
}
|
||||
|
||||
/** Read a file through `FileSystem` and sniff its media type from magic bytes, then the extension. */
|
||||
export const file = (path: string, options?: AssetOptions): Effect.Effect<Asset, AIError, FileSystem.FileSystem> =>
|
||||
Effect.gen(function* () {
|
||||
const fs = yield* FileSystem.FileSystem
|
||||
const data = yield* fs
|
||||
.readFile(path)
|
||||
.pipe(Effect.mapError((cause) => invalid(`Failed to read media file ${path}`, cause)))
|
||||
return bytes(data, detectMediaType(data) ?? extensionMediaType(path), options)
|
||||
})
|
||||
|
||||
/** Materialize an asset and write its bytes through `FileSystem`. */
|
||||
export const write = (asset: Asset, path: string): Effect.Effect<void, AIError, FileSystem.FileSystem | Service> =>
|
||||
Effect.gen(function* () {
|
||||
const fs = yield* FileSystem.FileSystem
|
||||
const data = yield* asset.bytes()
|
||||
yield* fs
|
||||
.writeFile(path, data)
|
||||
.pipe(Effect.mapError((cause) => invalid(`Failed to write media file ${path}`, cause)))
|
||||
})
|
||||
@@ -0,0 +1,186 @@
|
||||
import { Effect, Layer, ManagedRuntime, Stream } from "effect"
|
||||
import type { AwaitOptions, Generation, Snapshot } from "./generation.js"
|
||||
import { Image, ImageModel, ImageRequest, type ImageRequestInput } from "./image.js"
|
||||
import { ImageClient } from "./image-client.js"
|
||||
import { LLM } from "./index.js"
|
||||
import { LLMClient } from "./route/client.js"
|
||||
import { RequestExecutor } from "./route/executor.js"
|
||||
import { LanguageModel, LLMRequest } from "./schema/index.js"
|
||||
import type { RequestInput } from "./llm.js"
|
||||
import { Speech, SpeechModel, SpeechRequest, type SpeechRequestInput } from "./speech.js"
|
||||
import { SpeechClient } from "./speech-client.js"
|
||||
import {
|
||||
Transcription,
|
||||
TranscriptionModel,
|
||||
TranscriptionRequest,
|
||||
type TranscriptionOptions,
|
||||
type TranscriptionRequestInput,
|
||||
} from "./transcription.js"
|
||||
import { TranscriptionClient } from "./transcription-client.js"
|
||||
import { Video, VideoModel, VideoRequest, type VideoOptions, type VideoRequestInput } from "./video.js"
|
||||
import { VideoClient } from "./video-client.js"
|
||||
|
||||
/**
|
||||
* Promise-first entrypoint for scripts and non-Effect callers. One `ManagedRuntime` hosts the LLM, image, video, speech,
|
||||
* and transcription clients over a request executor; every method runs the corresponding Effect API and rethrows
|
||||
* `AIError` unchanged.
|
||||
*/
|
||||
export interface Options {
|
||||
/** Executor layer; defaults to `RequestExecutor.fetchLayer`. Inject a recorder or middleware here. */
|
||||
readonly layer?: Layer.Layer<RequestExecutor.Service>
|
||||
}
|
||||
|
||||
export interface RunOptions {
|
||||
readonly signal?: AbortSignal
|
||||
}
|
||||
|
||||
export type Services =
|
||||
| Layer.Success<typeof LLMClient.layer>
|
||||
| Layer.Success<typeof ImageClient.layer>
|
||||
| Layer.Success<typeof VideoClient.layer>
|
||||
| Layer.Success<typeof SpeechClient.layer>
|
||||
| Layer.Success<typeof TranscriptionClient.layer>
|
||||
| RequestExecutor.Service
|
||||
|
||||
/** Promise view of a `Generation`: its snapshot plus `await`, `refresh`, and `cancel` returning promises. */
|
||||
export type GenerationHandle<Response> = Snapshot & {
|
||||
/** Serializable JSON; pass it back to `resume` from another process. */
|
||||
readonly token: unknown
|
||||
readonly await: (options?: AwaitOptions & RunOptions) => Promise<Response>
|
||||
readonly refresh: (options?: RunOptions) => Promise<GenerationHandle<Response>>
|
||||
readonly cancel: (options?: RunOptions) => Promise<void>
|
||||
}
|
||||
|
||||
const abortEffect = (signal: AbortSignal | undefined) =>
|
||||
signal === undefined
|
||||
? Effect.never
|
||||
: Effect.callback<void>((resume) => {
|
||||
if (signal.aborted) {
|
||||
resume(Effect.void)
|
||||
return
|
||||
}
|
||||
const onAbort = () => resume(Effect.void)
|
||||
signal.addEventListener("abort", onAbort, { once: true })
|
||||
return Effect.sync(() => signal.removeEventListener("abort", onAbort))
|
||||
})
|
||||
|
||||
export const make = (options: Options = {}) => {
|
||||
const runtime = ManagedRuntime.make(
|
||||
Layer.mergeAll(
|
||||
LLMClient.layer,
|
||||
ImageClient.layer,
|
||||
VideoClient.layer,
|
||||
SpeechClient.layer,
|
||||
TranscriptionClient.layer,
|
||||
).pipe(Layer.provideMerge(options.layer ?? RequestExecutor.fetchLayer)),
|
||||
)
|
||||
|
||||
/** Run any package Effect (for example `asset.bytes()`) inside this runtime. */
|
||||
const run = <A, E>(effect: Effect.Effect<A, E, Services>, options?: RunOptions) =>
|
||||
runtime.runPromise(effect, { signal: options?.signal })
|
||||
|
||||
const iterate = <A, E>(stream: Stream.Stream<A, E, Services>, options?: RunOptions): AsyncIterable<A> =>
|
||||
Stream.toAsyncIterable(
|
||||
Stream.unwrap(
|
||||
runtime.contextEffect.pipe(
|
||||
Effect.map(
|
||||
(context): Stream.Stream<A, E> =>
|
||||
stream.pipe(Stream.interruptWhen(abortEffect(options?.signal)), Stream.provideContext(context)),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
const handle = <Response>(generation: Generation<Response>): GenerationHandle<Response> => ({
|
||||
...generation.snapshot,
|
||||
token: generation.token,
|
||||
await: (options) => run(generation.await({ poll: options?.poll }), options),
|
||||
refresh: (options) => run(generation.refresh(), options).then(handle),
|
||||
cancel: (options) => run(generation.cancel(), options),
|
||||
})
|
||||
|
||||
// The typed `generate`/`stream` overloads take a concrete input or a request, not the union; normalize once here.
|
||||
const llmRequest = (input: RequestInput | LLMRequest) => (input instanceof LLMRequest ? input : LLM.request(input))
|
||||
const imageRequest = (input: ImageRequestInput | ImageRequest) =>
|
||||
input instanceof ImageRequest ? input : Image.request(input)
|
||||
const videoRequest = (input: VideoRequestInput | VideoRequest) =>
|
||||
input instanceof VideoRequest ? input : Video.request(input)
|
||||
const speechRequest = (input: SpeechRequestInput | SpeechRequest) =>
|
||||
input instanceof SpeechRequest ? input : Speech.request(input)
|
||||
const transcriptionRequest = (input: TranscriptionRequestInput | TranscriptionRequest) =>
|
||||
input instanceof TranscriptionRequest ? input : Transcription.request(input)
|
||||
|
||||
return {
|
||||
run,
|
||||
llm: {
|
||||
request: LLM.request,
|
||||
generate: <const Model extends LanguageModel>(input: RequestInput<Model> | LLMRequest, options?: RunOptions) =>
|
||||
run(LLM.generate(llmRequest(input)), options),
|
||||
stream: <const Model extends LanguageModel>(input: RequestInput<Model> | LLMRequest, options?: RunOptions) =>
|
||||
iterate(LLM.stream(llmRequest(input)), options),
|
||||
},
|
||||
image: {
|
||||
request: Image.request,
|
||||
generate: <const Model extends ImageModel>(
|
||||
input: ImageRequestInput<Model> | ImageRequest,
|
||||
options?: RunOptions,
|
||||
) => run(Image.generate(imageRequest(input)), options),
|
||||
stream: <const Model extends ImageModel>(input: ImageRequestInput<Model> | ImageRequest, options?: RunOptions) =>
|
||||
iterate(Image.stream(imageRequest(input)), options),
|
||||
},
|
||||
video: {
|
||||
request: Video.request,
|
||||
start: <const Model extends VideoModel>(input: VideoRequestInput<Model> | VideoRequest, options?: RunOptions) =>
|
||||
run(Video.start(videoRequest(input)), options).then(handle),
|
||||
generate: <const Model extends VideoModel>(
|
||||
input: VideoRequestInput<Model> | VideoRequest,
|
||||
options?: AwaitOptions & RunOptions,
|
||||
) => run(Video.generate(videoRequest(input), { poll: options?.poll }), options),
|
||||
resume: <Options extends VideoOptions>(model: VideoModel<Options>, token: unknown, options?: RunOptions) =>
|
||||
run(Video.resume(model, token), options).then(handle),
|
||||
stream: <const Model extends VideoModel>(
|
||||
input: VideoRequestInput<Model> | VideoRequest,
|
||||
options?: AwaitOptions & RunOptions,
|
||||
) => iterate(Video.stream(videoRequest(input), { poll: options?.poll }), options),
|
||||
},
|
||||
speech: {
|
||||
request: Speech.request,
|
||||
generate: <const Model extends SpeechModel>(
|
||||
input: SpeechRequestInput<Model> | SpeechRequest,
|
||||
options?: RunOptions,
|
||||
) => run(Speech.generate(speechRequest(input)), options),
|
||||
stream: <const Model extends SpeechModel>(
|
||||
input: SpeechRequestInput<Model> | SpeechRequest,
|
||||
options?: RunOptions,
|
||||
) => iterate(Speech.stream(speechRequest(input)), options),
|
||||
},
|
||||
transcription: {
|
||||
request: Transcription.request,
|
||||
generate: <const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model> | TranscriptionRequest,
|
||||
options?: AwaitOptions & RunOptions,
|
||||
) => run(Transcription.generate(transcriptionRequest(input), { poll: options?.poll }), options),
|
||||
stream: <const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model> | TranscriptionRequest,
|
||||
options?: AwaitOptions & RunOptions,
|
||||
) => iterate(Transcription.stream(transcriptionRequest(input), { poll: options?.poll }), options),
|
||||
start: <const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model> | TranscriptionRequest,
|
||||
options?: RunOptions,
|
||||
) => run(Transcription.start(transcriptionRequest(input)), options).then(handle),
|
||||
resume: <Options extends TranscriptionOptions>(
|
||||
model: TranscriptionModel<Options>,
|
||||
token: unknown,
|
||||
options?: RunOptions,
|
||||
) => run(Transcription.resume(model, token), options).then(handle),
|
||||
},
|
||||
dispose: () => runtime.dispose(),
|
||||
}
|
||||
}
|
||||
|
||||
export type Client = ReturnType<typeof make>
|
||||
|
||||
/** Default client over `RequestExecutor.fetchLayer` for scripts; the runtime builds its layer on first use. */
|
||||
export const ai = make()
|
||||
|
||||
export * as AI from "./promise.js"
|
||||
@@ -658,7 +658,7 @@ const lowerMedia = Effect.fn("AnthropicMessages.lowerMedia")(function* (
|
||||
part: MediaPart,
|
||||
breakpoints?: Cache.Breakpoints,
|
||||
) {
|
||||
const mime = part.mediaType.toLowerCase()
|
||||
const mime = part.media.mediaType.toLowerCase()
|
||||
const cacheControlValue = breakpoints ? cacheControl(breakpoints, part.cache) : undefined
|
||||
const fileId = fileIdFromMetadata(part.metadata)
|
||||
|
||||
@@ -687,9 +687,9 @@ const lowerMedia = Effect.fn("AnthropicMessages.lowerMedia")(function* (
|
||||
} satisfies AnthropicDocumentBlock
|
||||
}
|
||||
|
||||
const rawString = typeof part.data === "string" ? part.data.trim() : undefined
|
||||
const rawString = ProviderShared.mediaUrl(part.media)?.trim()
|
||||
// SDK URL sources: URLImageSource:3817 / URLPDFSource:3823 {type:"url", url}
|
||||
if (rawString && isHttpUrl(rawString) && !rawString.startsWith("data:")) {
|
||||
if (rawString && isHttpUrl(rawString)) {
|
||||
if (mime.startsWith("image/"))
|
||||
return {
|
||||
type: "image" as const,
|
||||
@@ -714,20 +714,11 @@ const lowerMedia = Effect.fn("AnthropicMessages.lowerMedia")(function* (
|
||||
} satisfies AnthropicDocumentBlock
|
||||
}
|
||||
|
||||
const media = yield* ProviderShared.requireInlineMedia("Anthropic Messages", part.media)
|
||||
|
||||
// SDK PlainTextSource:2716 {type:"text", media_type:"text/plain", data}
|
||||
if (mime === "text/plain") {
|
||||
const textData =
|
||||
typeof part.data !== "string"
|
||||
? Buffer.from(part.data).toString("utf8")
|
||||
: part.data.startsWith("data:")
|
||||
? (() => {
|
||||
const comma = part.data.indexOf(",")
|
||||
const payload = comma >= 0 ? part.data.slice(comma + 1) : part.data
|
||||
return part.data.includes(";base64")
|
||||
? Buffer.from(payload, "base64").toString("utf8")
|
||||
: decodeURIComponent(payload)
|
||||
})()
|
||||
: part.data
|
||||
const textData = Buffer.from(media.base64, "base64").toString("utf8")
|
||||
return {
|
||||
type: "document" as const,
|
||||
source: { type: "text" as const, media_type: "text/plain" as const, data: textData },
|
||||
@@ -742,7 +733,6 @@ const lowerMedia = Effect.fn("AnthropicMessages.lowerMedia")(function* (
|
||||
} satisfies AnthropicDocumentBlock
|
||||
}
|
||||
|
||||
const media = ProviderShared.normalizeMedia(part)
|
||||
if (media.mime === "application/pdf")
|
||||
return {
|
||||
type: "document" as const,
|
||||
@@ -761,7 +751,7 @@ const lowerMedia = Effect.fn("AnthropicMessages.lowerMedia")(function* (
|
||||
: { citations: citationsFromMetadata(part.metadata)! }),
|
||||
} satisfies AnthropicDocumentBlock
|
||||
if (!media.mime.startsWith("image/"))
|
||||
return yield* invalid(`Anthropic Messages does not support media type ${part.mediaType}`)
|
||||
return yield* invalid(`Anthropic Messages does not support media type ${part.media.mediaType}`)
|
||||
return {
|
||||
type: "image" as const,
|
||||
source: {
|
||||
@@ -780,7 +770,7 @@ const lowerMedia = Effect.fn("AnthropicMessages.lowerMedia")(function* (
|
||||
// content instead of JSON-stringifying base64 into a prompt string.
|
||||
const lowerToolResultContentItem = Effect.fnUntraced(function* (item: Tool.Content) {
|
||||
if (item.type === "text") return { type: "text" as const, text: item.text } satisfies AnthropicTextBlock
|
||||
return yield* lowerMedia({ type: "media", mediaType: item.mime, data: item.uri, filename: item.name })
|
||||
return yield* lowerMedia(ProviderShared.toolFileMedia(item))
|
||||
})
|
||||
|
||||
const lowerToolResultContent = Effect.fnUntraced(function* (part: ToolResultPart) {
|
||||
|
||||
@@ -0,0 +1,211 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import type { Status } from "../generation.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { TranscriptionModel, TranscriptionResponse, type TranscriptionRequestFor } from "../transcription.js"
|
||||
import { ProviderShared, optionalNull } from "./shared.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "assemblyai-transcription"
|
||||
const NAME = "AssemblyAI"
|
||||
const PROVIDER = ProviderID.make("assemblyai")
|
||||
export const DEFAULT_BASE_URL = "https://api.assemblyai.com"
|
||||
export const PATH = "/v2/transcript"
|
||||
export const UPLOAD_PATH = "/v2/upload"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type AssemblyAITranscriptionOptions = {
|
||||
readonly keyterms_prompt?: ReadonlyArray<string>
|
||||
readonly punctuate?: boolean
|
||||
readonly format_text?: boolean
|
||||
readonly disfluencies?: boolean
|
||||
readonly filter_profanity?: boolean
|
||||
readonly temperature?: number
|
||||
readonly speaker_options?: { readonly min_speakers_expected?: number; readonly max_speakers_expected?: number }
|
||||
readonly language_detection_options?: {
|
||||
readonly expected_languages?: ReadonlyArray<string>
|
||||
readonly fallback_language?: string
|
||||
readonly code_switching?: boolean
|
||||
}
|
||||
readonly speech_models?: ReadonlyArray<"universal-3-5-pro" | "universal-2" | (string & {})>
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = TranscriptionRequestFor<AssemblyAITranscriptionOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Token and response schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const Token = Schema.Struct({ transcriptID: Schema.String })
|
||||
export type Token = Schema.Schema.Type<typeof Token>
|
||||
|
||||
const Upload = Schema.Struct({ upload_url: Schema.String })
|
||||
|
||||
/** Word and utterance times are milliseconds. */
|
||||
const Transcript = Schema.Struct({
|
||||
id: Schema.String,
|
||||
status: Schema.String,
|
||||
text: optionalNull(Schema.String),
|
||||
words: optionalNull(
|
||||
Schema.Array(
|
||||
Schema.Struct({
|
||||
text: Schema.String,
|
||||
start: Schema.Number,
|
||||
end: Schema.Number,
|
||||
confidence: optionalNull(Schema.Number),
|
||||
speaker: optionalNull(Schema.String),
|
||||
}),
|
||||
),
|
||||
),
|
||||
utterances: optionalNull(
|
||||
Schema.Array(
|
||||
Schema.Struct({
|
||||
text: Schema.String,
|
||||
start: Schema.Number,
|
||||
end: Schema.Number,
|
||||
speaker: optionalNull(Schema.String),
|
||||
}),
|
||||
),
|
||||
),
|
||||
language_code: optionalNull(Schema.String),
|
||||
audio_duration: optionalNull(Schema.Number),
|
||||
speech_model_used: optionalNull(Schema.String),
|
||||
error: optionalNull(Schema.String),
|
||||
})
|
||||
|
||||
const STATUS = {
|
||||
queued: "queued",
|
||||
processing: "running",
|
||||
completed: "completed",
|
||||
error: "failed",
|
||||
} as const satisfies Record<string, Status>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeUpload = MediaProtocol.decodeJson(ADAPTER, NAME, Upload)
|
||||
|
||||
/** `/v2/transcript` only takes a URL, so inline audio is uploaded to `/v2/upload` first. */
|
||||
const prepare = Effect.fn("AssemblyAITranscription.prepare")(function* (request: Request, send: MediaProtocol.Send) {
|
||||
if (request.audio.source.type !== "bytes" && request.audio.source.type !== "base64") return request
|
||||
const audio = yield* MediaInput.inlineBytes(ADAPTER, request.audio)
|
||||
const uploaded = yield* send(UPLOAD_PATH, MediaProtocol.binary(audio, "application/octet-stream")).pipe(
|
||||
Effect.flatMap(decodeUpload),
|
||||
)
|
||||
return { ...request, audio: Media.url(uploaded.value.upload_url, { mediaType: request.audio.mediaType }) }
|
||||
})
|
||||
|
||||
const fromRequest = Effect.fn("AssemblyAITranscription.fromRequest")(function* (request: Request) {
|
||||
const audio = yield* ProviderShared.mediaReference(request.audio, PROVIDER, NAME)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
audio_url: audio.value,
|
||||
speech_models: [request.model.id],
|
||||
language_code: request.language,
|
||||
language_detection: request.language === undefined ? true : undefined,
|
||||
prompt: request.prompt,
|
||||
// Turn-level `utterances`, the only segments AssemblyAI returns, require speaker labels.
|
||||
speaker_labels: request.diarize === true || request.timestamps === "segment" ? true : undefined,
|
||||
speakers_expected: request.speakers,
|
||||
},
|
||||
request.providerOptions,
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeTranscript = MediaProtocol.decodeJson(ADAPTER, NAME, Transcript)
|
||||
|
||||
const decodeStart = Effect.fn("AssemblyAITranscription.decodeStart")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const output = yield* decodeTranscript(response)
|
||||
const status = yield* MediaProtocol.status(STATUS, output.value.status, output)
|
||||
return { token: { transcriptID: output.value.id }, snapshot: { id: output.value.id, status } }
|
||||
})
|
||||
|
||||
const decodeStatus = Effect.fn("AssemblyAITranscription.decodeStatus")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeTranscript(response)
|
||||
const status = yield* MediaProtocol.status(STATUS, output.value.status, output)
|
||||
return { id: context.token.transcriptID, status }
|
||||
})
|
||||
|
||||
const seconds = (milliseconds: number) => milliseconds / 1000
|
||||
|
||||
const decodeResult = Effect.fn("AssemblyAITranscription.decodeResult")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeTranscript(response)
|
||||
const transcript = output.value
|
||||
const status = yield* MediaProtocol.status(STATUS, transcript.status, output)
|
||||
const error = transcript.error ?? undefined
|
||||
if (status === "failed")
|
||||
return yield* output.ended("failed", `${NAME} transcription failed${error === undefined ? "" : `: ${error}`}`)
|
||||
if (status !== "completed")
|
||||
return yield* output.invalid(`${NAME} transcript ${context.token.transcriptID} has not finished`)
|
||||
const duration = transcript.audio_duration ?? undefined
|
||||
return new TranscriptionResponse({
|
||||
text: transcript.text ?? "",
|
||||
segments: transcript.utterances?.map((utterance) => ({
|
||||
text: utterance.text,
|
||||
startSeconds: seconds(utterance.start),
|
||||
endSeconds: seconds(utterance.end),
|
||||
speaker: utterance.speaker ?? undefined,
|
||||
})),
|
||||
words: transcript.words?.map((word) => ({
|
||||
text: word.text,
|
||||
startSeconds: seconds(word.start),
|
||||
endSeconds: seconds(word.end),
|
||||
speaker: word.speaker ?? undefined,
|
||||
confidence: word.confidence ?? undefined,
|
||||
})),
|
||||
language: transcript.language_code?.toLowerCase(),
|
||||
durationSeconds: duration,
|
||||
usage: duration === undefined ? undefined : { type: "seconds", seconds: duration },
|
||||
providerMetadata: {
|
||||
assemblyai: { transcriptId: transcript.id, speechModel: transcript.speech_model_used ?? undefined },
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const transcriptPath = (token: Token) => `${PATH}/${token.transcriptID}`
|
||||
|
||||
export const protocol = MediaProtocol.queued<Request, TranscriptionResponse, Token>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
token: Token,
|
||||
start: { prepare, body: { from: fromRequest }, decode: decodeStart },
|
||||
status: { path: transcriptPath, decode: decodeStatus },
|
||||
result: { path: transcriptPath, decode: decodeResult },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
TranscriptionModel.fromRoute<AssemblyAITranscriptionOptions, Token>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH },
|
||||
input,
|
||||
)
|
||||
|
||||
export const AssemblyAITranscription = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -28,6 +28,7 @@ import { Lifecycle } from "./utils/lifecycle.js"
|
||||
import { MistralToolID } from "./utils/mistral-tool-id.js"
|
||||
import { ToolSchemaProjection } from "./utils/tool-schema.js"
|
||||
import { ToolStream } from "./utils/tool-stream.js"
|
||||
import { concatBytes } from "../utils/bytes.js"
|
||||
|
||||
const ADAPTER = "bedrock-converse"
|
||||
|
||||
@@ -303,15 +304,7 @@ const lowerToolResultContent = Effect.fn("BedrockConverse.lowerToolResultContent
|
||||
content.push({ text: item.text })
|
||||
continue
|
||||
}
|
||||
const media = yield* BedrockMedia.lower(
|
||||
{
|
||||
type: "media",
|
||||
mediaType: item.mime,
|
||||
data: item.uri,
|
||||
filename: item.name,
|
||||
},
|
||||
documentNames,
|
||||
)
|
||||
const media = yield* BedrockMedia.lower(ProviderShared.toolFileMedia(item), documentNames)
|
||||
content.push(...media)
|
||||
}
|
||||
return content
|
||||
@@ -532,14 +525,7 @@ interface ParserState {
|
||||
readonly reasoningRedactedContent: Readonly<Record<number, ReadonlyArray<Uint8Array>>>
|
||||
}
|
||||
|
||||
const encodeRedactedContent = (chunks: ReadonlyArray<Uint8Array>) => {
|
||||
const bytes = new Uint8Array(chunks.reduce((total, chunk) => total + chunk.length, 0))
|
||||
chunks.reduce((offset, chunk) => {
|
||||
bytes.set(chunk, offset)
|
||||
return offset + chunk.length
|
||||
}, 0)
|
||||
return Encoding.encodeBase64(bytes)
|
||||
}
|
||||
const encodeRedactedContent = (chunks: ReadonlyArray<Uint8Array>) => Encoding.encodeBase64(concatBytes(chunks))
|
||||
|
||||
const step = (state: ParserState, event: BedrockEvent) =>
|
||||
Effect.gen(function* () {
|
||||
|
||||
@@ -0,0 +1,195 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { classifyProviderFailure } from "../provider-error.js"
|
||||
import { Framing } from "../route/framing.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { AIError, ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { SpeechModel, type SpeechEvent, type SpeechRequestFor } from "../speech.js"
|
||||
import { ProviderShared, optionalNull } from "./shared.js"
|
||||
import { SpeechStream } from "./utils/speech-stream.js"
|
||||
|
||||
const ADAPTER = "cartesia-speech"
|
||||
const NAME = "Cartesia"
|
||||
const PROVIDER = ProviderID.make("cartesia")
|
||||
export const DEFAULT_BASE_URL = "https://api.cartesia.ai"
|
||||
export const API_VERSION = "2026-08-14"
|
||||
export const BYTES_PATH = "/tts/bytes"
|
||||
export const SSE_PATH = "/tts/sse"
|
||||
const DEFAULT_SAMPLE_RATE = 44100
|
||||
const DEFAULT_BIT_RATE = 128000
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type CartesiaSpeechString<Known extends string> = Known | (string & {})
|
||||
|
||||
export type CartesiaEncoding = SpeechStream.PcmEncoding
|
||||
|
||||
export type CartesiaSpeechOptions = {
|
||||
readonly sampleRate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000
|
||||
readonly bitRate?: 32000 | 64000 | 96000 | 128000 | 192000
|
||||
readonly encoding?: CartesiaEncoding
|
||||
readonly generation_config?: {
|
||||
readonly volume?: number
|
||||
readonly emotion?: CartesiaSpeechString<"neutral" | "calm" | "angry" | "content" | "sad" | "scared">
|
||||
}
|
||||
readonly pronunciation_dict_id?: string
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = SpeechRequestFor<CartesiaSpeechOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Streaming event schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** `phoneme_timestamps` and future record types are ignored. */
|
||||
const SseEvent = Schema.Struct({
|
||||
type: Schema.String,
|
||||
data: Schema.optional(Schema.Uint8ArrayFromBase64),
|
||||
word_timestamps: Schema.optional(
|
||||
Schema.Struct({
|
||||
words: Schema.Array(Schema.String),
|
||||
start: Schema.Array(Schema.Number),
|
||||
end: Schema.Array(Schema.Number),
|
||||
}),
|
||||
),
|
||||
status_code: Schema.optional(Schema.Number),
|
||||
title: Schema.optional(Schema.String),
|
||||
message: Schema.optional(Schema.String),
|
||||
error_code: optionalNull(Schema.String),
|
||||
})
|
||||
|
||||
const decodeEvent = MediaProtocol.decodeFrame(ADAPTER, NAME, SseEvent)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface State extends SpeechStream.Audio {
|
||||
readonly done: boolean
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Timestamps exist only on the SSE endpoint, so a `generate` that asks for them collects an SSE stream. */
|
||||
const usesSse = (request: MediaProtocol.Addressed<Request>) => request.mode === "stream" || request.timestamps === true
|
||||
|
||||
const CONTAINERS: Readonly<Record<string, "raw" | "wav" | "mp3">> = { pcm: "raw", wav: "wav", mp3: "mp3" }
|
||||
|
||||
const outputFormat = Effect.fn("CartesiaSpeech.outputFormat")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
const sse = usesSse(request)
|
||||
const format = request.format ?? (sse ? "pcm" : "mp3")
|
||||
const container = CONTAINERS[format]
|
||||
if (container === undefined)
|
||||
return yield* SpeechStream.unsupportedFormat(
|
||||
PROVIDER,
|
||||
ADAPTER,
|
||||
`${NAME} supports the pcm, wav, and mp3 formats, not "${format}"`,
|
||||
)
|
||||
if (sse && container !== "raw")
|
||||
return yield* SpeechStream.unsupportedFormat(
|
||||
PROVIDER,
|
||||
ADAPTER,
|
||||
`${NAME} streams and timestamps only raw PCM; request format "pcm" instead of "${format}"`,
|
||||
)
|
||||
const sampleRate = request.providerOptions?.sampleRate ?? DEFAULT_SAMPLE_RATE
|
||||
if (container === "mp3")
|
||||
return { container, sample_rate: sampleRate, bit_rate: request.providerOptions?.bitRate ?? DEFAULT_BIT_RATE }
|
||||
return { container, encoding: request.providerOptions?.encoding ?? "pcm_s16le", sample_rate: sampleRate }
|
||||
})
|
||||
|
||||
const fromRequest = Effect.fn("CartesiaSpeech.fromRequest")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
const voice = SpeechStream.voiceID(request.voice)
|
||||
if (voice === undefined)
|
||||
return yield* ProviderShared.invalidRequest(`${NAME} requires a voice id; pass it as \`voice\``)
|
||||
const { sampleRate: _sampleRate, bitRate: _bitRate, encoding: _encoding, ...native } = request.providerOptions ?? {}
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model_id: request.model.id,
|
||||
transcript: request.text,
|
||||
voice,
|
||||
output_format: yield* outputFormat(request),
|
||||
language: request.language,
|
||||
generation_config: request.speed === undefined ? undefined : { speed: request.speed },
|
||||
add_timestamps: request.timestamps === true ? true : undefined,
|
||||
},
|
||||
native,
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const onEvent = Effect.fn("CartesiaSpeech.onEvent")(function* (state: State, frame: string) {
|
||||
const event = yield* decodeEvent(frame)
|
||||
if (event.type === "chunk" && event.data !== undefined) return SpeechStream.delta(state, event.data)
|
||||
if (event.type === "timestamps" && event.word_timestamps !== undefined) {
|
||||
const words = event.word_timestamps
|
||||
return [state, SpeechStream.timestamps(words.words, words.start, words.end)] as const
|
||||
}
|
||||
if (event.type === "done") return [{ ...state, done: true }, []] as const
|
||||
if (event.type === "error")
|
||||
return yield* new AIError({
|
||||
reason: classifyProviderFailure({
|
||||
message: `${NAME} stream failed${event.title === undefined ? "" : ` (${event.title})`}: ${event.message ?? "unknown error"}`,
|
||||
status: event.status_code,
|
||||
rawBody: frame,
|
||||
}),
|
||||
})
|
||||
return [state, []] as const
|
||||
})
|
||||
|
||||
const finish = Effect.fn("CartesiaSpeech.finish")(function* (
|
||||
state: State,
|
||||
context: MediaProtocol.ResponseContext<Request>,
|
||||
) {
|
||||
if (usesSse(context.request) && !state.done) return yield* MediaProtocol.incomplete(ADAPTER)
|
||||
const format = yield* outputFormat(context.request)
|
||||
return yield* SpeechStream.finish(
|
||||
ADAPTER,
|
||||
state,
|
||||
format.container === "raw"
|
||||
? SpeechStream.pcm(format.encoding, format.sample_rate)
|
||||
: SpeechStream.container(format.container, format.sample_rate),
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, SpeechEvent, string | Uint8Array, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["instructions"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes, context) => (usesSse(context.request) ? Framing.sse.frame(bytes) : bytes),
|
||||
initial: () => ({ chunks: [], done: false }),
|
||||
step: SpeechStream.step(onEvent),
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
SpeechModel.fromRoute<CartesiaSpeechOptions, string | Uint8Array, State>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
headers: { "Cartesia-Version": API_VERSION },
|
||||
path: ({ request }) => (usesSse(request) ? SSE_PATH : BYTES_PATH),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const CartesiaSpeech = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,144 @@
|
||||
import { Effect } from "effect"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { SpeechModel, type SpeechEvent, type SpeechRequestFor } from "../speech.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
import { SpeechStream } from "./utils/speech-stream.js"
|
||||
|
||||
const ADAPTER = "deepgram-speech"
|
||||
const NAME = "Deepgram"
|
||||
const PROVIDER = ProviderID.make("deepgram")
|
||||
export const DEFAULT_BASE_URL = "https://api.deepgram.com"
|
||||
export const PATH = "/v1/speak"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type DeepgramSpeechString<Known extends string> = Known | (string & {})
|
||||
|
||||
export type DeepgramEncoding = DeepgramSpeechString<"linear16" | "mulaw" | "alaw" | "mp3" | "opus" | "flac" | "aac">
|
||||
|
||||
export type DeepgramSpeechOptions = {
|
||||
readonly encoding?: DeepgramEncoding
|
||||
readonly container?: DeepgramSpeechString<"wav" | "ogg" | "none">
|
||||
readonly sampleRate?: number
|
||||
readonly bitRate?: number
|
||||
readonly mip_opt_out?: boolean
|
||||
readonly tag?: string
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = SpeechRequestFor<DeepgramSpeechOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type State = SpeechStream.Audio
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const FORMATS: Readonly<Record<string, { readonly encoding: string; readonly container?: string }>> = {
|
||||
mp3: { encoding: "mp3" },
|
||||
wav: { encoding: "linear16", container: "wav" },
|
||||
pcm: { encoding: "linear16", container: "none" },
|
||||
opus: { encoding: "opus" },
|
||||
flac: { encoding: "flac" },
|
||||
aac: { encoding: "aac" },
|
||||
}
|
||||
|
||||
const audioFormat = (request: Request) => {
|
||||
const format = request.format === undefined ? undefined : FORMATS[request.format]
|
||||
return {
|
||||
encoding: request.providerOptions?.encoding ?? format?.encoding,
|
||||
container: request.providerOptions?.container ?? format?.container,
|
||||
}
|
||||
}
|
||||
|
||||
const queryParameters = (request: Request) => {
|
||||
const { encoding: _encoding, container: _container, sampleRate, bitRate, ...native } = request.providerOptions ?? {}
|
||||
return MediaInput.query(ADAPTER, {
|
||||
...native,
|
||||
model: request.model.id,
|
||||
...audioFormat(request),
|
||||
sample_rate: sampleRate,
|
||||
bit_rate: bitRate,
|
||||
speed: request.speed,
|
||||
})
|
||||
}
|
||||
|
||||
const fromRequest = Effect.fn("DeepgramSpeech.fromRequest")(function* (request: Request) {
|
||||
if (
|
||||
request.format !== undefined &&
|
||||
FORMATS[request.format] === undefined &&
|
||||
request.providerOptions?.encoding === undefined
|
||||
)
|
||||
return yield* SpeechStream.unsupportedFormat(
|
||||
PROVIDER,
|
||||
ADAPTER,
|
||||
`${NAME} has no encoding for format "${request.format}"; pass providerOptions.encoding`,
|
||||
)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords({ text: request.text }, request.http?.body) ?? {},
|
||||
yield* queryParameters(request),
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const HEADERLESS_ENCODINGS: Readonly<Record<string, SpeechStream.PcmEncoding>> = {
|
||||
linear16: "pcm_s16le",
|
||||
mulaw: "pcm_mulaw",
|
||||
alaw: "pcm_alaw",
|
||||
}
|
||||
|
||||
const finish = (state: State, context: MediaProtocol.ResponseContext<Request>) => {
|
||||
const headers = context.http.headers
|
||||
const mediaType = headers["content-type"]
|
||||
const format = audioFormat(context.request)
|
||||
const encoding = HEADERLESS_ENCODINGS[format.encoding ?? ""]
|
||||
const requestID = headers["dg-request-id"]
|
||||
const modelName = headers["dg-model-name"]
|
||||
return SpeechStream.finish(ADAPTER, state, {
|
||||
...(format.container === "none" && encoding !== undefined
|
||||
? SpeechStream.pcm(encoding, SpeechStream.sampleRate(mediaType), mediaType)
|
||||
: // Deepgram's default encoding is MP3; WAV is a container around any encoding.
|
||||
{ mediaType, info: { format: format.container === "wav" ? "wav" : (format.encoding ?? "mp3") } }),
|
||||
usage: SpeechStream.headerUsage("characters", headers["dg-char-count"]),
|
||||
providerMetadata:
|
||||
requestID === undefined && modelName === undefined
|
||||
? undefined
|
||||
: { deepgram: { requestId: requestID, modelName } },
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, SpeechEvent, Uint8Array, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["voice", "language", "instructions", "timestamps"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes) => bytes,
|
||||
initial: () => ({ chunks: [] }),
|
||||
step: (state, frame) => Effect.succeed(SpeechStream.delta(state, frame)),
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
SpeechModel.fromRoute<DeepgramSpeechOptions, Uint8Array, State>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH },
|
||||
input,
|
||||
)
|
||||
|
||||
export const DeepgramSpeech = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,191 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { TranscriptionModel, TranscriptionResponse, type TranscriptionRequestFor } from "../transcription.js"
|
||||
import { ProviderShared } from "./shared.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "deepgram-transcription"
|
||||
const NAME = "Deepgram"
|
||||
const PROVIDER = ProviderID.make("deepgram")
|
||||
export const DEFAULT_BASE_URL = "https://api.deepgram.com"
|
||||
export const PATH = "/v1/listen"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type DeepgramTranscriptionOptions = {
|
||||
readonly smart_format?: boolean
|
||||
readonly punctuate?: boolean
|
||||
readonly paragraphs?: boolean
|
||||
readonly utterances?: boolean
|
||||
readonly detect_language?: boolean | ReadonlyArray<string>
|
||||
readonly keyterm?: ReadonlyArray<string>
|
||||
readonly diarize_model?: "latest" | "v1" | "v2" | (string & {})
|
||||
readonly filler_words?: boolean
|
||||
readonly numerals?: boolean
|
||||
readonly mip_opt_out?: boolean
|
||||
readonly tag?: string | ReadonlyArray<string>
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = TranscriptionRequestFor<DeepgramTranscriptionOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Response schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const Word = Schema.Struct({
|
||||
word: Schema.String,
|
||||
start: Schema.Number,
|
||||
end: Schema.Number,
|
||||
confidence: Schema.optional(Schema.Number),
|
||||
speaker: Schema.optional(Schema.Number),
|
||||
punctuated_word: Schema.optional(Schema.String),
|
||||
})
|
||||
|
||||
const ListenResponse = Schema.Struct({
|
||||
metadata: Schema.optional(
|
||||
Schema.Struct({ request_id: Schema.optional(Schema.String), duration: Schema.optional(Schema.Number) }),
|
||||
),
|
||||
results: Schema.Struct({
|
||||
channels: Schema.Array(
|
||||
Schema.Struct({
|
||||
alternatives: Schema.optional(
|
||||
Schema.Array(Schema.Struct({ transcript: Schema.String, words: Schema.optional(Schema.Array(Word)) })),
|
||||
),
|
||||
detected_language: Schema.optional(Schema.String),
|
||||
}),
|
||||
),
|
||||
utterances: Schema.optional(
|
||||
Schema.Array(
|
||||
Schema.Struct({
|
||||
start: Schema.Number,
|
||||
end: Schema.Number,
|
||||
transcript: Schema.String,
|
||||
speaker: Schema.optional(Schema.Number),
|
||||
words: Schema.optional(Schema.Array(Word)),
|
||||
}),
|
||||
),
|
||||
),
|
||||
}),
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const query = (request: Request) =>
|
||||
MediaInput.query(
|
||||
ADAPTER,
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
smart_format: true,
|
||||
language: request.language,
|
||||
// Deepgram assumes English unless asked to detect, unlike the other routes' auto-detection.
|
||||
detect_language: request.language === undefined ? true : undefined,
|
||||
// `diarize=true` is deprecated in favor of choosing a diarization model.
|
||||
diarize_model: request.diarize === true ? "latest" : undefined,
|
||||
utterances: request.diarize === true || request.timestamps === "segment" ? true : undefined,
|
||||
},
|
||||
request.providerOptions,
|
||||
) ?? {},
|
||||
)
|
||||
|
||||
const fromRequest = Effect.fn("DeepgramTranscription.fromRequest")(function* (request: Request) {
|
||||
const url = ProviderShared.mediaUrl(request.audio)
|
||||
if (url !== undefined)
|
||||
return MediaProtocol.json(mergeJsonRecords({ url }, request.http?.body) ?? {}, yield* query(request))
|
||||
if (request.http?.body !== undefined)
|
||||
return yield* ProviderShared.invalidRequest(`${NAME} sends inline audio as the raw body, so http.body cannot apply`)
|
||||
const audio = yield* MediaInput.inlineBytes(ADAPTER, request.audio)
|
||||
return MediaProtocol.binary(audio, request.audio.mediaType, yield* query(request))
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeListen = MediaProtocol.decodeJson(ADAPTER, NAME, ListenResponse)
|
||||
|
||||
const speaker = (value: number | undefined) => (value === undefined ? undefined : String(value))
|
||||
|
||||
const wordText = (word: typeof Word.Type) => word.punctuated_word ?? word.word
|
||||
|
||||
// Utterances split on pauses, not speakers: the v2 diarizer labels a whole utterance with one speaker even when its
|
||||
// words change speaker, so segments split each utterance at speaker changes.
|
||||
const speakerTurns = (words: ReadonlyArray<typeof Word.Type>) =>
|
||||
words.reduce<Array<Array<typeof Word.Type>>>((turns, word) => {
|
||||
const last = turns.at(-1)
|
||||
if (last === undefined || last[0].speaker !== word.speaker) return [...turns, [word]]
|
||||
last.push(word)
|
||||
return turns
|
||||
}, [])
|
||||
|
||||
const decodeResponse = Effect.fn("DeepgramTranscription.decodeResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const output = yield* decodeListen(response)
|
||||
const channel = output.value.results.channels[0]
|
||||
const alternative = channel?.alternatives?.[0]
|
||||
if (alternative === undefined) return yield* output.invalid(`${NAME} returned no transcript`)
|
||||
const duration = output.value.metadata?.duration
|
||||
const requestID = output.value.metadata?.request_id
|
||||
return new TranscriptionResponse({
|
||||
text: alternative.transcript,
|
||||
segments: output.value.results.utterances?.flatMap((utterance) =>
|
||||
utterance.words === undefined || utterance.words.length === 0
|
||||
? [
|
||||
{
|
||||
text: utterance.transcript,
|
||||
startSeconds: utterance.start,
|
||||
endSeconds: utterance.end,
|
||||
speaker: speaker(utterance.speaker),
|
||||
},
|
||||
]
|
||||
: speakerTurns(utterance.words).map((turn) => ({
|
||||
text: turn.map(wordText).join(" "),
|
||||
startSeconds: turn[0].start,
|
||||
endSeconds: turn[turn.length - 1].end,
|
||||
speaker: speaker(turn[0].speaker),
|
||||
})),
|
||||
),
|
||||
words: alternative.words?.map((word) => ({
|
||||
text: wordText(word),
|
||||
startSeconds: word.start,
|
||||
endSeconds: word.end,
|
||||
speaker: speaker(word.speaker),
|
||||
confidence: word.confidence,
|
||||
})),
|
||||
language: channel?.detected_language?.toLowerCase(),
|
||||
durationSeconds: duration,
|
||||
usage: duration === undefined ? undefined : { type: "seconds", seconds: duration },
|
||||
providerMetadata: requestID === undefined ? undefined : { deepgram: { requestId: requestID } },
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.inline<Request, TranscriptionResponse>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["prompt", "speakers"],
|
||||
body: { from: fromRequest },
|
||||
response: { decode: decodeResponse },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
TranscriptionModel.fromRoute<DeepgramTranscriptionOptions>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH },
|
||||
input,
|
||||
)
|
||||
|
||||
export const DeepgramTranscription = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,218 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { Framing } from "../route/framing.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { SpeechModel, type SpeechEvent, type SpeechRequestFor } from "../speech.js"
|
||||
import { ProviderShared, optionalNull } from "./shared.js"
|
||||
import { SpeechStream } from "./utils/speech-stream.js"
|
||||
|
||||
const ADAPTER = "elevenlabs-speech"
|
||||
const NAME = "ElevenLabs"
|
||||
const PROVIDER = ProviderID.make("elevenlabs")
|
||||
export const DEFAULT_BASE_URL = "https://api.elevenlabs.io"
|
||||
export const PATH = "/v1/text-to-speech"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type ElevenLabsSpeechString<Known extends string> = Known | (string & {})
|
||||
|
||||
export type ElevenLabsOutputFormat = ElevenLabsSpeechString<
|
||||
| "mp3_22050_32"
|
||||
| "mp3_24000_48"
|
||||
| "mp3_44100_32"
|
||||
| "mp3_44100_64"
|
||||
| "mp3_44100_96"
|
||||
| "mp3_44100_128"
|
||||
| "mp3_44100_192"
|
||||
| "pcm_8000"
|
||||
| "pcm_16000"
|
||||
| "pcm_22050"
|
||||
| "pcm_24000"
|
||||
| "pcm_32000"
|
||||
| "pcm_44100"
|
||||
| "pcm_48000"
|
||||
| "wav_8000"
|
||||
| "wav_16000"
|
||||
| "wav_22050"
|
||||
| "wav_24000"
|
||||
| "wav_32000"
|
||||
| "wav_44100"
|
||||
| "wav_48000"
|
||||
| "ulaw_8000"
|
||||
| "alaw_8000"
|
||||
| "opus_48000_32"
|
||||
| "opus_48000_64"
|
||||
| "opus_48000_96"
|
||||
| "opus_48000_128"
|
||||
| "opus_48000_192"
|
||||
>
|
||||
|
||||
export type ElevenLabsSpeechOptions = {
|
||||
readonly outputFormat?: ElevenLabsOutputFormat
|
||||
readonly voice_settings?: {
|
||||
readonly stability?: number
|
||||
readonly similarity_boost?: number
|
||||
readonly style?: number
|
||||
readonly use_speaker_boost?: boolean
|
||||
}
|
||||
readonly seed?: number
|
||||
readonly apply_text_normalization?: ElevenLabsSpeechString<"auto" | "on" | "off">
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = SpeechRequestFor<ElevenLabsSpeechOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Streaming event schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const Alignment = Schema.Struct({
|
||||
characters: Schema.Array(Schema.String),
|
||||
character_start_times_seconds: Schema.Array(Schema.Number),
|
||||
character_end_times_seconds: Schema.Array(Schema.Number),
|
||||
})
|
||||
|
||||
const TimestampedAudio = Schema.Struct({
|
||||
audio_base64: Schema.Uint8ArrayFromBase64,
|
||||
alignment: optionalNull(Alignment),
|
||||
})
|
||||
|
||||
const decodeRecord = MediaProtocol.decodeFrame(ADAPTER, NAME, TimestampedAudio)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type State = SpeechStream.Audio
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const OUTPUT_FORMATS: Readonly<Record<string, string>> = {
|
||||
mp3: "mp3_44100_128",
|
||||
pcm: "pcm_24000",
|
||||
wav: "wav_24000",
|
||||
opus: "opus_48000_64",
|
||||
}
|
||||
|
||||
/** WAV is served only by the non-streaming endpoints. */
|
||||
const outputFormat = Effect.fn("ElevenLabsSpeech.outputFormat")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
const format = request.providerOptions?.outputFormat ?? OUTPUT_FORMATS[request.format ?? "mp3"]
|
||||
if (format === undefined)
|
||||
return yield* SpeechStream.unsupportedFormat(
|
||||
PROVIDER,
|
||||
ADAPTER,
|
||||
`${NAME} has no default output format for "${request.format}"; pass providerOptions.outputFormat`,
|
||||
)
|
||||
if (request.mode === "stream" && format.startsWith("wav_"))
|
||||
return yield* SpeechStream.unsupportedFormat(
|
||||
PROVIDER,
|
||||
ADAPTER,
|
||||
`${NAME} streams mp3, pcm, opus, ulaw, and alaw but not "${format}"; use generate for WAV`,
|
||||
)
|
||||
return format
|
||||
})
|
||||
|
||||
const fromRequest = Effect.fn("ElevenLabsSpeech.fromRequest")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
if (request.voice === undefined)
|
||||
return yield* ProviderShared.invalidRequest(`${NAME} requires a voice id; pass it as \`voice\``)
|
||||
const { outputFormat: _outputFormat, ...native } = request.providerOptions ?? {}
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
text: request.text,
|
||||
model_id: request.model.id,
|
||||
language_code: request.language,
|
||||
voice_settings: request.speed === undefined ? undefined : { speed: request.speed },
|
||||
},
|
||||
native,
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
{ output_format: yield* outputFormat(request) },
|
||||
)
|
||||
})
|
||||
|
||||
const path = (request: MediaProtocol.Addressed<Request>) =>
|
||||
`${PATH}/${encodeURIComponent(SpeechStream.voiceID(request.voice) ?? "")}${request.mode === "stream" ? "/stream" : ""}${
|
||||
request.timestamps === true ? "/with-timestamps" : ""
|
||||
}`
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const onRecord = Effect.fn("ElevenLabsSpeech.onRecord")(function* (state: State, frame: string) {
|
||||
const record = yield* decodeRecord(frame)
|
||||
const [next, events] = SpeechStream.delta(state, record.audio_base64)
|
||||
const alignment = record.alignment
|
||||
if (!alignment) return [next, events] as const
|
||||
return [
|
||||
next,
|
||||
[
|
||||
...events,
|
||||
...SpeechStream.timestamps(
|
||||
alignment.characters,
|
||||
alignment.character_start_times_seconds,
|
||||
alignment.character_end_times_seconds,
|
||||
),
|
||||
],
|
||||
] as const
|
||||
})
|
||||
|
||||
const PCM_CODECS: Readonly<Record<string, SpeechStream.PcmEncoding>> = {
|
||||
pcm: "pcm_s16le",
|
||||
ulaw: "pcm_mulaw",
|
||||
alaw: "pcm_alaw",
|
||||
}
|
||||
|
||||
const describeOutput = (format: string) => {
|
||||
const [codec = format, rate] = format.split("_")
|
||||
const sampleRate = rate === undefined ? undefined : Number(rate)
|
||||
const encoding = PCM_CODECS[codec]
|
||||
return encoding === undefined ? SpeechStream.container(codec, sampleRate) : SpeechStream.pcm(encoding, sampleRate)
|
||||
}
|
||||
|
||||
const finish = Effect.fn("ElevenLabsSpeech.finish")(function* (
|
||||
state: State,
|
||||
context: MediaProtocol.ResponseContext<Request>,
|
||||
) {
|
||||
const requestID = context.http.headers["request-id"]
|
||||
return yield* SpeechStream.finish(ADAPTER, state, {
|
||||
...describeOutput(yield* outputFormat(context.request)),
|
||||
// `character-cost` is billed credits, not a character count (3 for 20 characters on `eleven_flash_v2_5`).
|
||||
usage: SpeechStream.headerUsage("credits", context.http.headers["character-cost"]),
|
||||
providerMetadata: requestID === undefined ? undefined : { elevenlabs: { requestId: requestID } },
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, SpeechEvent, string | Uint8Array, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["instructions"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes, context) => {
|
||||
if (context.request.timestamps !== true) return bytes
|
||||
return context.request.mode === "stream" ? Framing.lines.frame(bytes) : Framing.document.frame(bytes)
|
||||
},
|
||||
initial: () => ({ chunks: [] }),
|
||||
step: SpeechStream.step(onRecord),
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
SpeechModel.fromRoute<ElevenLabsSpeechOptions, string | Uint8Array, State>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: ({ request }) => path(request) },
|
||||
input,
|
||||
)
|
||||
|
||||
export const ElevenLabsSpeech = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,198 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import type { Status } from "../generation.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { VideoModel, VideoResponse, type VideoRequestFor } from "../video.js"
|
||||
import { ProviderShared, optionalNull } from "./shared.js"
|
||||
|
||||
const ADAPTER = "fal-video"
|
||||
const NAME = "fal Video"
|
||||
const PROVIDER = ProviderID.make("fal")
|
||||
export const DEFAULT_BASE_URL = "https://queue.fal.run"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type FalVideoString<Known extends string> = Known | (string & {})
|
||||
|
||||
/**
|
||||
* Provider-native input. fal video endpoints are model-specific: `duration` is a string enum whose values differ per
|
||||
* model (`"8s"` for Veo, `"5"` for Kling), and last-frame fields are named per model (`end_image_url`,
|
||||
* `last_frame_url`, `tail_image_url`), so those pass through here instead of lowering from common fields.
|
||||
*/
|
||||
export type FalVideoOptions = {
|
||||
readonly duration?: FalVideoString<"4s" | "6s" | "8s" | "5" | "10">
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = VideoRequestFor<FalVideoOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Token and response schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** fal hands back absolute follow-up URLs on submit; they are authoritative for status, result, and cancel. */
|
||||
export const Token = Schema.Struct({
|
||||
requestID: Schema.String,
|
||||
statusURL: Schema.String,
|
||||
responseURL: Schema.String,
|
||||
cancelURL: Schema.String,
|
||||
})
|
||||
export type Token = Schema.Schema.Type<typeof Token>
|
||||
|
||||
const StartResponse = Schema.Struct({
|
||||
request_id: Schema.String,
|
||||
status_url: Schema.String,
|
||||
response_url: Schema.String,
|
||||
cancel_url: Schema.String,
|
||||
queue_position: optionalNull(Schema.Number),
|
||||
})
|
||||
|
||||
const QueueStatus = Schema.Struct({
|
||||
status: Schema.String,
|
||||
queue_position: optionalNull(Schema.Number),
|
||||
error: optionalNull(Schema.Unknown),
|
||||
})
|
||||
|
||||
const QueueResult = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
video: Schema.Struct({
|
||||
url: Schema.String,
|
||||
content_type: optionalNull(Schema.String),
|
||||
file_name: optionalNull(Schema.String),
|
||||
file_size: optionalNull(Schema.Number),
|
||||
}),
|
||||
seed: optionalNull(Schema.Number),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
)
|
||||
|
||||
const STATUS = {
|
||||
IN_QUEUE: "queued",
|
||||
IN_PROGRESS: "running",
|
||||
COMPLETED: "completed",
|
||||
} as const satisfies Record<string, Status>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// fal accepts public URLs and data URIs; there is no provider file handle to forward.
|
||||
const mediaUrl = (asset: Media.Asset) =>
|
||||
ProviderShared.mediaReference(asset, undefined, NAME).pipe(Effect.map((reference) => reference.value))
|
||||
|
||||
const fromRequest = Effect.fn("FalVideo.fromRequest")(function* (request: Request) {
|
||||
if (request.frames?.last !== undefined)
|
||||
return yield* ProviderShared.unsupportedOperation({
|
||||
operation: "video.frames.last",
|
||||
provider: PROVIDER,
|
||||
route: ADAPTER,
|
||||
message: `${NAME} names the last frame per model; pass it through providerOptions (e.g. end_image_url) instead of frames.last`,
|
||||
})
|
||||
const imageUrl = request.frames?.first === undefined ? undefined : yield* mediaUrl(request.frames.first)
|
||||
const videoUrl = request.video === undefined ? undefined : yield* mediaUrl(request.video)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
prompt: request.prompt,
|
||||
negative_prompt: request.negativePrompt,
|
||||
seed: request.seed,
|
||||
aspect_ratio: request.aspectRatio,
|
||||
resolution: request.resolution,
|
||||
generate_audio: request.audio,
|
||||
image_url: imageUrl,
|
||||
video_url: videoUrl,
|
||||
},
|
||||
request.providerOptions,
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeStart = MediaProtocol.decodeStarted(ADAPTER, NAME, StartResponse, (value) => ({
|
||||
token: {
|
||||
requestID: value.request_id,
|
||||
statusURL: value.status_url,
|
||||
responseURL: value.response_url,
|
||||
cancelURL: value.cancel_url,
|
||||
},
|
||||
snapshot: { id: value.request_id, status: "queued", position: value.queue_position ?? undefined },
|
||||
}))
|
||||
|
||||
const decodeQueueStatus = MediaProtocol.decodeJson(ADAPTER, NAME, QueueStatus)
|
||||
const decodeQueueResult = MediaProtocol.decodeJson(ADAPTER, NAME, QueueResult)
|
||||
|
||||
const decodeStatus = Effect.fn("FalVideo.decodeStatus")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeQueueStatus(response)
|
||||
const decoded = output.value
|
||||
const status = yield* MediaProtocol.status(STATUS, decoded.status, output)
|
||||
// fal reports request failures as COMPLETED with an `error`; the response endpoint carries the details.
|
||||
const failed = status === "completed" && decoded.error !== undefined && decoded.error !== null
|
||||
return {
|
||||
id: context.token.requestID,
|
||||
status: failed ? "failed" : status,
|
||||
position: status === "queued" ? (decoded.queue_position ?? undefined) : undefined,
|
||||
}
|
||||
})
|
||||
|
||||
const decodeResult = Effect.fn("FalVideo.decodeResult")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeQueueResult(response)
|
||||
const { video, seed, ...rest } = output.value
|
||||
return new VideoResponse({
|
||||
videos: [Media.url(video.url, { mediaType: video.content_type ?? "video/mp4" })],
|
||||
providerMetadata: {
|
||||
fal: {
|
||||
requestId: context.token.requestID,
|
||||
seed: seed ?? undefined,
|
||||
fileName: video.file_name ?? undefined,
|
||||
fileSize: video.file_size ?? undefined,
|
||||
...rest,
|
||||
},
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.queued<Request, VideoResponse, Token>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
token: Token,
|
||||
unsupported: ["n", "durationSeconds", "references"],
|
||||
start: { body: { from: fromRequest }, decode: decodeStart },
|
||||
status: { path: (token) => token.statusURL, decode: decodeStatus },
|
||||
result: { path: (token) => token.responseURL, decode: decodeResult },
|
||||
cancel: { method: "PUT", path: (token) => token.cancelURL },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
VideoModel.fromRoute<FalVideoOptions, Token>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => `/${request.model.id}`,
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const FalVideo = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -20,7 +20,9 @@ import {
|
||||
type ToolDefinition,
|
||||
} from "../schema/index.js"
|
||||
import { classifyProviderFailure } from "../provider-error.js"
|
||||
import { Media } from "../media.js"
|
||||
import { JsonObject, knownString, lenient, optionalArray, optionalNull, ProviderShared } from "./shared.js"
|
||||
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js"
|
||||
import { GeminiToolSchema } from "./utils/gemini-tool-schema.js"
|
||||
import { Lifecycle } from "./utils/lifecycle.js"
|
||||
import { ToolSchemaProjection } from "./utils/tool-schema.js"
|
||||
@@ -74,9 +76,18 @@ const GeminiInlineDataPart = Schema.Struct({
|
||||
mimeType: Schema.String,
|
||||
data: Schema.String,
|
||||
}),
|
||||
thoughtSignature: optionalNull(Schema.String),
|
||||
})
|
||||
type GeminiInlineDataPart = Schema.Schema.Type<typeof GeminiInlineDataPart>
|
||||
|
||||
/** Gemini Files API reference; the only remote input Gemini accepts. */
|
||||
const GeminiFileDataPart = Schema.Struct({
|
||||
fileData: Schema.Struct({
|
||||
mimeType: Schema.String,
|
||||
fileUri: Schema.String,
|
||||
}),
|
||||
})
|
||||
|
||||
const GeminiFunctionCallPart = Schema.Struct({
|
||||
functionCall: Schema.Struct({
|
||||
id: optionalNull(Schema.String),
|
||||
@@ -98,6 +109,7 @@ const GeminiFunctionResponsePart = Schema.Struct({
|
||||
const GeminiContentPart = Schema.Union([
|
||||
GeminiTextPart,
|
||||
GeminiInlineDataPart,
|
||||
GeminiFileDataPart,
|
||||
GeminiFunctionCallPart,
|
||||
GeminiFunctionResponsePart,
|
||||
])
|
||||
@@ -294,10 +306,9 @@ const lowerToolConfig = (toolChoice: NonNullable<LLMRequest["toolChoice"]>) =>
|
||||
tool: (name) => ({ functionCallingConfig: { mode: "ANY" as const, allowedFunctionNames: [name] } }),
|
||||
})
|
||||
|
||||
const lowerUserPart = Effect.fn("Gemini.lowerUserPart")(function* (part: TextPart | MediaPart) {
|
||||
const lowerContentPart = Effect.fn("Gemini.lowerContentPart")(function* (part: TextPart | MediaPart) {
|
||||
if (part.type === "text") return { text: part.text }
|
||||
const media = ProviderShared.normalizeMedia(part)
|
||||
return { inlineData: { mimeType: media.mime, data: media.base64 } }
|
||||
return yield* GeminiGenerateContent.mediaPart("Gemini", part.media)
|
||||
})
|
||||
|
||||
const providerMetadata = (key: string, metadata: Record<string, unknown>): ProviderMetadata => ({ [key]: metadata })
|
||||
@@ -344,7 +355,7 @@ const lowerMessages = Effect.fn("Gemini.lowerMessages")(function* (request: LLMR
|
||||
for (const part of message.content) {
|
||||
if (!ProviderShared.supportsContent(part, ["text", "media"]))
|
||||
return yield* ProviderShared.unsupportedContent("Gemini", "user", ["text", "media"])
|
||||
parts.push(yield* lowerUserPart(part))
|
||||
parts.push(yield* lowerContentPart(part))
|
||||
}
|
||||
contents.push({ role: "user", parts })
|
||||
continue
|
||||
@@ -355,12 +366,23 @@ const lowerMessages = Effect.fn("Gemini.lowerMessages")(function* (request: LLMR
|
||||
// Parallel Gemini 3 calls may carry one signature on the first call; unsigned sibling calls are valid.
|
||||
let hasSignedToolCall = false
|
||||
for (const part of message.content) {
|
||||
if (!ProviderShared.supportsContent(part, ["text", "reasoning", "tool-call"]))
|
||||
return yield* ProviderShared.unsupportedContent("Gemini", "assistant", ["text", "reasoning", "tool-call"])
|
||||
if (!ProviderShared.supportsContent(part, ["text", "reasoning", "tool-call", "media"]))
|
||||
return yield* ProviderShared.unsupportedContent("Gemini", "assistant", [
|
||||
"text",
|
||||
"reasoning",
|
||||
"tool-call",
|
||||
"media",
|
||||
])
|
||||
if (part.type === "text") {
|
||||
parts.push({ text: part.text, thoughtSignature: thoughtSignature(part.providerMetadata, metadataKey) })
|
||||
continue
|
||||
}
|
||||
// Generated images replay as model-role inline data so multi-turn image editing keeps the prior output.
|
||||
if (part.type === "media") {
|
||||
const lowered = yield* lowerContentPart(part)
|
||||
parts.push({ ...lowered, thoughtSignature: thoughtSignature(part.providerMetadata, metadataKey) })
|
||||
continue
|
||||
}
|
||||
if (part.type === "reasoning") {
|
||||
parts.push({
|
||||
text: part.text,
|
||||
@@ -410,7 +432,7 @@ const lowerMessages = Effect.fn("Gemini.lowerMessages")(function* (request: LLMR
|
||||
const media: GeminiInlineDataPart[] = []
|
||||
for (const item of content) {
|
||||
if (item.type === "text") continue
|
||||
const value = ProviderShared.normalizeToolFile(item)
|
||||
const value = yield* ProviderShared.requireInlineMedia("Gemini", ProviderShared.toolFileMedia(item).media)
|
||||
media.push({ inlineData: { mimeType: value.mime, data: value.base64 } })
|
||||
}
|
||||
if (legacyToolMedia && media.length > 0) (pendingMedia ??= []).push(...media)
|
||||
@@ -662,6 +684,19 @@ const step = (state: ParserState, event: GeminiEvent) => {
|
||||
// each block kind must retain the signature attached to its own parts.
|
||||
if (signature !== undefined && "thought" in part && part.thought) reasoningSignature = signature
|
||||
else if (signature !== undefined && "text" in part) textSignature = signature
|
||||
// Image-capable Gemini models return generated images as inline data parts; surface them as first-class output.
|
||||
if ("inlineData" in part) {
|
||||
lifecycle = Lifecycle.stepStart(lifecycle, events)
|
||||
events.push(
|
||||
LLMEvent.media({
|
||||
media: Media.base64(part.inlineData.data, part.inlineData.mimeType),
|
||||
providerMetadata: signature
|
||||
? providerMetadata(state.providerMetadataKey, { thoughtSignature: signature })
|
||||
: undefined,
|
||||
}),
|
||||
)
|
||||
continue
|
||||
}
|
||||
if ("text" in part && part.text.length > 0) {
|
||||
if (part.thought) {
|
||||
if (textId !== undefined) {
|
||||
|
||||
@@ -1,40 +1,36 @@
|
||||
import { Effect, Encoding, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import {
|
||||
GeneratedImage,
|
||||
ImageModel,
|
||||
ImageResponse,
|
||||
type ImageInput,
|
||||
type ImageRequestFor,
|
||||
type ImageRoute,
|
||||
} from "../image.js"
|
||||
import { Auth, type Definition as AuthDefinition } from "../route/auth.js"
|
||||
import { AIError, Usage, mergeHttpOptions, mergeJsonRecords, type HttpOptions } from "../schema/index.js"
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { ProviderShared } from "./shared.js"
|
||||
import { ImageInputs } from "./utils/image-input.js"
|
||||
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "google-images"
|
||||
const NAME = "Google Images"
|
||||
const PROVIDER = ProviderID.make("google")
|
||||
export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type GoogleImageString<Known extends string> = Known | (string & {})
|
||||
|
||||
/** Provider-native options. Common fields (`aspectRatio`, `seed`, `images`) live on the request. */
|
||||
export type GoogleImageOptions = {
|
||||
readonly aspectRatio?: GoogleImageString<
|
||||
"1:1" | "2:3" | "3:2" | "3:4" | "4:3" | "4:5" | "5:4" | "9:16" | "16:9" | "21:9"
|
||||
>
|
||||
readonly imageSize?: GoogleImageString<"1K" | "2K" | "4K">
|
||||
readonly seed?: number
|
||||
readonly thinkingLevel?: GoogleImageString<"MINIMAL" | "LOW" | "MEDIUM" | "HIGH">
|
||||
readonly includeThoughts?: boolean
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type GoogleImageBody = Record<string, unknown> & {
|
||||
readonly contents: ReadonlyArray<{
|
||||
readonly role: "user"
|
||||
readonly parts: ReadonlyArray<Record<string, unknown>>
|
||||
}>
|
||||
readonly generationConfig: Record<string, unknown>
|
||||
}
|
||||
export type Request = ImageRequestFor<GoogleImageOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Response schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const GoogleUsage = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
@@ -85,30 +81,20 @@ const GoogleImageResponse = Schema.Struct({
|
||||
promptFeedback: Schema.optional(Schema.Unknown),
|
||||
})
|
||||
|
||||
export interface ModelInput {
|
||||
readonly id: string
|
||||
readonly auth: AuthDefinition
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const nativeOptions = (options: GoogleImageOptions | undefined) => {
|
||||
const { aspectRatio, imageSize, seed, thinkingLevel, includeThoughts, ...native } = options ?? {}
|
||||
const image = {
|
||||
aspectRatio,
|
||||
imageSize,
|
||||
}
|
||||
const thinkingConfig = {
|
||||
thinkingLevel,
|
||||
includeThoughts,
|
||||
}
|
||||
const generationConfig = (request: Request) => {
|
||||
const { imageSize, thinkingLevel, includeThoughts, ...native } = request.providerOptions ?? {}
|
||||
const imageConfig = { aspectRatio: request.aspectRatio, imageSize }
|
||||
const thinkingConfig = { thinkingLevel, includeThoughts }
|
||||
return (
|
||||
mergeJsonRecords(
|
||||
{
|
||||
responseModalities: ["IMAGE"],
|
||||
imageConfig: Object.values(image).some((value) => value !== undefined) ? image : undefined,
|
||||
seed,
|
||||
imageConfig: Object.values(imageConfig).some((value) => value !== undefined) ? imageConfig : undefined,
|
||||
seed: request.seed,
|
||||
thinkingConfig: Object.values(thinkingConfig).some((value) => value !== undefined) ? thinkingConfig : undefined,
|
||||
},
|
||||
native,
|
||||
@@ -116,176 +102,189 @@ const nativeOptions = (options: GoogleImageOptions | undefined) => {
|
||||
)
|
||||
}
|
||||
|
||||
const applyQuery = (url: string, query: Record<string, string> | undefined) => {
|
||||
if (!query) return url
|
||||
const next = new URL(url)
|
||||
Object.entries(query).forEach(([key, value]) => next.searchParams.set(key, value))
|
||||
return next.toString()
|
||||
}
|
||||
const fromRequest = Effect.fn("GoogleImages.fromRequest")(function* (request: Request) {
|
||||
if (request.n !== undefined && request.n > 1)
|
||||
return yield* ProviderShared.unsupportedOperation({
|
||||
operation: "image.n",
|
||||
provider: PROVIDER,
|
||||
route: ADAPTER,
|
||||
message: `${NAME} generates one image per request; call it once per image instead of n=${request.n}`,
|
||||
})
|
||||
const parts = yield* Effect.forEach(request.images ?? [], (image) => GeminiGenerateContent.mediaPart(NAME, image))
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
contents: [{ role: "user", parts: [{ text: request.prompt }, ...parts] }],
|
||||
generationConfig: generationConfig(request),
|
||||
},
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
export const model = (input: ModelInput) => {
|
||||
const route: ImageRoute<GoogleImageOptions> = {
|
||||
id: ADAPTER,
|
||||
generate: Effect.fn("GoogleImages.generate")(function* (request: ImageRequestFor<GoogleImageOptions>, execute) {
|
||||
const imageParts = yield* Effect.forEach(request.images ?? [], googleImagePart)
|
||||
const http = mergeHttpOptions(request.model.http, request.http)
|
||||
const requestBody = mergeJsonRecords(
|
||||
{
|
||||
contents: [{ role: "user", parts: [{ text: request.prompt }, ...imageParts] }],
|
||||
generationConfig: nativeOptions(request.options),
|
||||
},
|
||||
http?.body,
|
||||
) as GoogleImageBody
|
||||
const text = ProviderShared.encodeJson(requestBody)
|
||||
const url = applyQuery(
|
||||
`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}/models/${request.model.id}:generateContent`,
|
||||
http?.query,
|
||||
)
|
||||
const headers = yield* Auth.toEffect(input.auth)({
|
||||
request,
|
||||
method: "POST",
|
||||
url,
|
||||
body: text,
|
||||
headers: Headers.fromInput({ ...input.headers, ...http?.headers }),
|
||||
})
|
||||
const response = yield* execute(
|
||||
HttpClientRequest.post(url).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(text, "application/json"),
|
||||
),
|
||||
)
|
||||
const output = yield* ProviderShared.imageResponse(ADAPTER, "Google Images", response)
|
||||
const decoded = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(GoogleImageResponse))(output.body).pipe(
|
||||
Effect.mapError((cause) => output.invalid("Google Images returned an invalid response", cause)),
|
||||
)
|
||||
const candidates = decoded.candidates ?? []
|
||||
const candidateMetadata = candidates.map((candidate, candidateIndex) => ({
|
||||
index: candidate.index ?? candidateIndex,
|
||||
finishReason: candidate.finishReason,
|
||||
finishMessage: candidate.finishMessage,
|
||||
safetyRatings: candidate.safetyRatings,
|
||||
citationMetadata: candidate.citationMetadata,
|
||||
groundingMetadata: candidate.groundingMetadata,
|
||||
parts: (candidate.content?.parts ?? []).map((part) =>
|
||||
part.inlineData === undefined
|
||||
? {
|
||||
type: "text",
|
||||
text: part.text,
|
||||
thought: part.thought,
|
||||
thoughtSignature: part.thoughtSignature,
|
||||
}
|
||||
: {
|
||||
type: "inlineData",
|
||||
mediaType: part.inlineData.mimeType,
|
||||
thought: part.thought,
|
||||
thoughtSignature: part.thoughtSignature,
|
||||
},
|
||||
),
|
||||
}))
|
||||
const encoded = candidates.flatMap((candidate, candidateIndex) =>
|
||||
(candidate.content?.parts ?? []).flatMap((part, partIndex) =>
|
||||
part.inlineData === undefined || part.thought === true
|
||||
? []
|
||||
: [{ candidate, candidateIndex, partIndex, inlineData: part.inlineData }],
|
||||
),
|
||||
)
|
||||
const images = yield* Effect.forEach(encoded, (item) =>
|
||||
Effect.fromResult(Encoding.decodeBase64(item.inlineData.data)).pipe(
|
||||
Effect.mapError((cause) =>
|
||||
output.invalid(
|
||||
`Google Images candidate ${item.candidateIndex} part ${item.partIndex} contains invalid base64 data`,
|
||||
cause,
|
||||
),
|
||||
),
|
||||
Effect.map(
|
||||
(data) =>
|
||||
new GeneratedImage({
|
||||
mediaType: item.inlineData.mimeType,
|
||||
data,
|
||||
providerMetadata: {
|
||||
google: {
|
||||
candidateIndex: item.candidate.index ?? item.candidateIndex,
|
||||
partIndex: item.partIndex,
|
||||
finishReason: item.candidate.finishReason,
|
||||
safetyRatings: item.candidate.safetyRatings,
|
||||
citationMetadata: item.candidate.citationMetadata,
|
||||
groundingMetadata: item.candidate.groundingMetadata,
|
||||
thoughtSignature: item.candidate.content?.parts[item.partIndex]?.thoughtSignature,
|
||||
},
|
||||
},
|
||||
}),
|
||||
),
|
||||
),
|
||||
)
|
||||
if (images.length === 0) {
|
||||
const finishReasons = candidates.flatMap((candidate) =>
|
||||
candidate.finishReason === undefined ? [] : [candidate.finishReason],
|
||||
)
|
||||
return yield* output.invalid(
|
||||
`Google Images returned no final images${
|
||||
finishReasons.length === 0 ? "" : ` (finish reasons: ${finishReasons.join(", ")})`
|
||||
}; inspect body for prompt feedback and candidate details`,
|
||||
)
|
||||
}
|
||||
const usage = decoded.usageMetadata
|
||||
const outputTokens =
|
||||
usage?.candidatesTokenCount === undefined
|
||||
? undefined
|
||||
: usage.candidatesTokenCount + (usage.thoughtsTokenCount ?? 0)
|
||||
return new ImageResponse({
|
||||
images,
|
||||
usage:
|
||||
usage === undefined
|
||||
? undefined
|
||||
: new Usage({
|
||||
inputTokens: usage.promptTokenCount,
|
||||
outputTokens,
|
||||
nonCachedInputTokens: ProviderShared.subtractTokens(
|
||||
usage.promptTokenCount,
|
||||
usage.cachedContentTokenCount,
|
||||
),
|
||||
cacheReadInputTokens: usage.cachedContentTokenCount,
|
||||
reasoningTokens: usage.thoughtsTokenCount,
|
||||
totalTokens: ProviderShared.totalTokens(usage.promptTokenCount, outputTokens, usage.totalTokenCount),
|
||||
providerMetadata: { google: usage },
|
||||
}),
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeResponse = Effect.fn("GoogleImages.decodeResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, GoogleImageResponse)(response)
|
||||
const decoded = output.value
|
||||
const candidates = decoded.candidates ?? []
|
||||
const candidateMetadata = candidates.map((candidate, candidateIndex) => ({
|
||||
index: candidate.index ?? candidateIndex,
|
||||
finishReason: candidate.finishReason,
|
||||
finishMessage: candidate.finishMessage,
|
||||
safetyRatings: candidate.safetyRatings,
|
||||
citationMetadata: candidate.citationMetadata,
|
||||
groundingMetadata: candidate.groundingMetadata,
|
||||
parts: (candidate.content?.parts ?? []).map((part) =>
|
||||
part.inlineData === undefined
|
||||
? { type: "text", text: part.text, thought: part.thought, thoughtSignature: part.thoughtSignature }
|
||||
: {
|
||||
type: "inlineData",
|
||||
mediaType: part.inlineData.mimeType,
|
||||
thought: part.thought,
|
||||
thoughtSignature: part.thoughtSignature,
|
||||
},
|
||||
),
|
||||
}))
|
||||
// Thought parts are drafts; only non-thought inline data is a final image.
|
||||
const encoded = candidates.flatMap((candidate, candidateIndex) =>
|
||||
(candidate.content?.parts ?? []).flatMap((part, partIndex) =>
|
||||
part.inlineData === undefined || part.thought === true
|
||||
? []
|
||||
: [
|
||||
{
|
||||
candidate,
|
||||
candidateIndex,
|
||||
partIndex,
|
||||
inlineData: part.inlineData,
|
||||
thoughtSignature: part.thoughtSignature,
|
||||
},
|
||||
],
|
||||
),
|
||||
)
|
||||
const images = yield* Effect.forEach(encoded, (item) =>
|
||||
MediaInput.decodedAsset(
|
||||
output.invalid,
|
||||
`${NAME} candidate ${item.candidateIndex} part ${item.partIndex}`,
|
||||
item.inlineData.data,
|
||||
item.inlineData.mimeType,
|
||||
{
|
||||
providerMetadata: {
|
||||
google: {
|
||||
modelVersion: decoded.modelVersion,
|
||||
responseId: decoded.responseId,
|
||||
promptFeedback: decoded.promptFeedback,
|
||||
candidates: candidateMetadata,
|
||||
candidateIndex: item.candidate.index ?? item.candidateIndex,
|
||||
partIndex: item.partIndex,
|
||||
finishReason: item.candidate.finishReason,
|
||||
safetyRatings: item.candidate.safetyRatings,
|
||||
citationMetadata: item.candidate.citationMetadata,
|
||||
groundingMetadata: item.candidate.groundingMetadata,
|
||||
thoughtSignature: item.thoughtSignature,
|
||||
},
|
||||
},
|
||||
})
|
||||
}),
|
||||
},
|
||||
),
|
||||
)
|
||||
if (images.length === 0) {
|
||||
const finishReasons = candidates.flatMap((candidate) =>
|
||||
candidate.finishReason === undefined ? [] : [candidate.finishReason],
|
||||
)
|
||||
return yield* output.invalid(
|
||||
`${NAME} returned no final images${
|
||||
finishReasons.length === 0 ? "" : ` (finish reasons: ${finishReasons.join(", ")})`
|
||||
}; inspect body for prompt feedback and candidate details`,
|
||||
)
|
||||
}
|
||||
return ImageModel.make<GoogleImageOptions>({ id: input.id, provider: "google", route, http: input.http })
|
||||
}
|
||||
// Candidates that stopped for a safety or policy reason are partial results, not a silent drop.
|
||||
const notices = [
|
||||
...(decoded.promptFeedback === undefined
|
||||
? []
|
||||
: [
|
||||
{
|
||||
type: "filtered" as const,
|
||||
message: `${NAME} reported prompt feedback`,
|
||||
providerMetadata: { google: { promptFeedback: decoded.promptFeedback } },
|
||||
},
|
||||
]),
|
||||
...candidates.flatMap((candidate, index) =>
|
||||
candidate.finishReason === undefined || candidate.finishReason === "STOP"
|
||||
? []
|
||||
: [
|
||||
{
|
||||
type: "filtered" as const,
|
||||
message: `${NAME} candidate ${candidate.index ?? index} finished with ${candidate.finishReason}${
|
||||
candidate.finishMessage === undefined ? "" : `: ${candidate.finishMessage}`
|
||||
}`,
|
||||
providerMetadata: {
|
||||
google: {
|
||||
candidateIndex: candidate.index ?? index,
|
||||
finishReason: candidate.finishReason,
|
||||
finishMessage: candidate.finishMessage,
|
||||
safetyRatings: candidate.safetyRatings,
|
||||
},
|
||||
},
|
||||
},
|
||||
],
|
||||
),
|
||||
]
|
||||
const usage = decoded.usageMetadata
|
||||
const outputTokens =
|
||||
usage?.candidatesTokenCount === undefined ? undefined : usage.candidatesTokenCount + (usage.thoughtsTokenCount ?? 0)
|
||||
return new ImageResponse({
|
||||
images,
|
||||
notices: notices.length === 0 ? undefined : notices,
|
||||
usage:
|
||||
usage === undefined
|
||||
? undefined
|
||||
: {
|
||||
type: "tokens",
|
||||
input: usage.promptTokenCount,
|
||||
output: outputTokens,
|
||||
total: ProviderShared.totalTokens(usage.promptTokenCount, outputTokens, usage.totalTokenCount),
|
||||
details: {
|
||||
reasoningTokens: usage.thoughtsTokenCount,
|
||||
cacheReadInputTokens: usage.cachedContentTokenCount,
|
||||
google: usage,
|
||||
},
|
||||
},
|
||||
providerMetadata: {
|
||||
google: {
|
||||
modelVersion: decoded.modelVersion,
|
||||
responseId: decoded.responseId,
|
||||
promptFeedback: decoded.promptFeedback,
|
||||
candidates: candidateMetadata,
|
||||
},
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const googleImagePart = (image: ImageInput): Effect.Effect<Record<string, unknown>, AIError> => {
|
||||
if (image.type === "bytes")
|
||||
return Effect.succeed({ inlineData: { mimeType: image.mediaType, data: Encoding.encodeBase64(image.data) } })
|
||||
if (image.type === "file-uri") return Effect.succeed({ fileData: { mimeType: image.mediaType, fileUri: image.uri } })
|
||||
if (image.type === "url")
|
||||
return ImageInputs.decodeDataUrl(image.url).pipe(
|
||||
Effect.flatMap((decoded) => {
|
||||
if (decoded === undefined)
|
||||
return Effect.fail(
|
||||
ImageInputs.invalid(
|
||||
"Google generateContent does not fetch public image URLs; use bytes, a data URL, or a Gemini file URI",
|
||||
),
|
||||
)
|
||||
return Effect.succeed({
|
||||
inlineData: { mimeType: decoded.mediaType, data: Encoding.encodeBase64(decoded.data) },
|
||||
})
|
||||
}),
|
||||
)
|
||||
return Effect.fail(
|
||||
ImageInputs.invalid("Google generateContent requires Gemini file URIs rather than provider file IDs"),
|
||||
export const protocol = MediaProtocol.inline<Request, ImageResponse>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["mask", "size", "format"],
|
||||
body: { from: fromRequest },
|
||||
response: { decode: decodeResponse },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
ImageModel.fromRoute<GoogleImageOptions>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => `/models/${request.model.id}:generateContent`,
|
||||
},
|
||||
input,
|
||||
)
|
||||
}
|
||||
|
||||
export const GoogleImages = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { SpeechModel, type SpeechEvent, type SpeechRequestFor } from "../speech.js"
|
||||
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js"
|
||||
import { SpeechStream } from "./utils/speech-stream.js"
|
||||
|
||||
const ADAPTER = "google-speech"
|
||||
const NAME = "Google Speech"
|
||||
const PROVIDER = ProviderID.make("google")
|
||||
export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"
|
||||
const DEFAULT_SAMPLE_RATE = 24000
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Style is directed in the text itself, and `speechConfig.multiSpeakerVoiceConfig` excludes `voice`. */
|
||||
export type GoogleSpeechOptions = {
|
||||
readonly temperature?: number
|
||||
readonly seed?: number
|
||||
readonly speechConfig?: {
|
||||
readonly multiSpeakerVoiceConfig?: {
|
||||
readonly speakerVoiceConfigs: ReadonlyArray<{
|
||||
readonly speaker: string
|
||||
readonly voiceConfig: { readonly prebuiltVoiceConfig: { readonly voiceName: string } }
|
||||
}>
|
||||
}
|
||||
}
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = SpeechRequestFor<GoogleSpeechOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Streaming event schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const GenerateContentChunk = GeminiGenerateContent.chunk(
|
||||
Schema.Struct({
|
||||
text: Schema.optional(Schema.String),
|
||||
inlineData: Schema.optional(Schema.Struct({ mimeType: Schema.String, data: Schema.Uint8ArrayFromBase64 })),
|
||||
}),
|
||||
)
|
||||
|
||||
const decodeChunk = MediaProtocol.decodeFrame(ADAPTER, NAME, GenerateContentChunk)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface State extends SpeechStream.Audio, GeminiGenerateContent.Metadata {
|
||||
readonly mimeType?: string
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const fromRequest = Effect.fn("GoogleSpeech.fromRequest")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
if (request.format !== undefined && request.format !== "pcm")
|
||||
return yield* SpeechStream.unsupportedFormat(
|
||||
PROVIDER,
|
||||
ADAPTER,
|
||||
`${NAME} only returns raw PCM; request format "pcm" or omit it, then wrap the samples yourself`,
|
||||
)
|
||||
const voiceName = SpeechStream.voiceID(request.voice)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
contents: [{ role: "user", parts: [{ text: request.text }] }],
|
||||
generationConfig: mergeJsonRecords(
|
||||
{
|
||||
responseModalities: ["AUDIO"],
|
||||
speechConfig: {
|
||||
voiceConfig: voiceName === undefined ? undefined : { prebuiltVoiceConfig: { voiceName } },
|
||||
languageCode: request.language,
|
||||
},
|
||||
},
|
||||
request.providerOptions,
|
||||
),
|
||||
},
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const step = Effect.fn("GoogleSpeech.step")(function* (state: State, frame: string) {
|
||||
const chunk = yield* decodeChunk(frame)
|
||||
const blocked = GeminiGenerateContent.blocked(NAME, chunk, frame)
|
||||
if (blocked !== undefined) return yield* blocked
|
||||
const audio = (chunk.candidates?.[0]?.content?.parts ?? []).flatMap((part) =>
|
||||
part.inlineData === undefined ? [] : [part.inlineData],
|
||||
)
|
||||
const next: State = { ...GeminiGenerateContent.track(state, chunk), mimeType: state.mimeType ?? audio[0]?.mimeType }
|
||||
return [next, audio.flatMap((part) => SpeechStream.delta(next, part.data)[1])] as const
|
||||
})
|
||||
|
||||
const finish = (state: State) => {
|
||||
const sampleRate = SpeechStream.sampleRate(state.mimeType) ?? DEFAULT_SAMPLE_RATE
|
||||
return SpeechStream.finish(ADAPTER, state, {
|
||||
...SpeechStream.pcm("pcm_s16le", sampleRate, state.mimeType ?? `audio/L16;codec=pcm;rate=${sampleRate}`),
|
||||
usage: GeminiGenerateContent.usage(state.usage),
|
||||
providerMetadata: GeminiGenerateContent.providerMetadata(state),
|
||||
detail: state.finishReason === undefined ? undefined : `finish reason: ${state.finishReason}`,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, SpeechEvent, string, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["instructions", "speed", "timestamps"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes, context) => GeminiGenerateContent.frames(bytes, context.request.mode),
|
||||
initial: () => ({ chunks: [] }),
|
||||
step,
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
SpeechModel.fromRoute<GoogleSpeechOptions, string, State>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
// Only `gemini-3.1-flash-tts-preview` and later stream; earlier TTS models reject `streamGenerateContent`.
|
||||
path: ({ request }) => GeminiGenerateContent.path(request.model.id, request.mode),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const GoogleSpeech = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,214 @@
|
||||
import { Effect, Schema, SchemaGetter } from "effect"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import {
|
||||
TranscriptionFinishEvent,
|
||||
TranscriptionModel,
|
||||
TranscriptionSegmentEvent,
|
||||
TranscriptionTextDeltaEvent,
|
||||
type TranscriptionRequestFor,
|
||||
type TranscriptionSegment,
|
||||
type TranscriptionWord,
|
||||
type TranscriptionEvent,
|
||||
} from "../transcription.js"
|
||||
import { ProviderShared } from "./shared.js"
|
||||
import { GeminiGenerateContent } from "./utils/gemini-generate-content.js"
|
||||
|
||||
const ADAPTER = "google-transcription"
|
||||
const NAME = "Google Transcription"
|
||||
const PROVIDER = ProviderID.make("google")
|
||||
export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Merged into `generationConfig`. The API rejects `customVocabulary` and `mode: "SMART"` alongside diarization or word
|
||||
* timestamps.
|
||||
*/
|
||||
export type GoogleTranscriptionOptions = {
|
||||
readonly audioTranscriptionConfig?: {
|
||||
readonly mode?: "VERBATIM" | "SMART" | (string & {})
|
||||
readonly customVocabulary?: ReadonlyArray<string>
|
||||
readonly languageCodes?: ReadonlyArray<string>
|
||||
}
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = TranscriptionRequestFor<GoogleTranscriptionOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Streaming event schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const Seconds = Schema.String.check(Schema.isPattern(/^\d+(\.\d+)?s$/)).pipe(
|
||||
Schema.decodeTo(Schema.Number, {
|
||||
decode: SchemaGetter.transform((value) => Number.parseFloat(value)),
|
||||
encode: SchemaGetter.transform((value) => `${value}s`),
|
||||
}),
|
||||
)
|
||||
|
||||
const AudioTranscription = Schema.Struct({
|
||||
text: Schema.String,
|
||||
speakerLabel: Schema.optional(Schema.String),
|
||||
words: Schema.optional(
|
||||
Schema.Array(
|
||||
Schema.Struct({
|
||||
word: Schema.String,
|
||||
startOffset: Seconds,
|
||||
endOffset: Seconds,
|
||||
speakerLabel: Schema.optional(Schema.String),
|
||||
}),
|
||||
),
|
||||
),
|
||||
})
|
||||
|
||||
const decodeChunk = MediaProtocol.decodeFrame(
|
||||
ADAPTER,
|
||||
NAME,
|
||||
GeminiGenerateContent.chunk(Schema.Struct({ audioTranscription: Schema.optional(AudioTranscription) })),
|
||||
)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface State extends GeminiGenerateContent.Metadata {
|
||||
readonly text: string
|
||||
readonly segments: Array<TranscriptionSegment>
|
||||
readonly words: Array<TranscriptionWord>
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const fromRequest = Effect.fn("GoogleTranscription.fromRequest")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
// General Gemini models ignore `audioTranscriptionConfig` and answer the audio conversationally.
|
||||
if (!request.model.id.includes("transcribe"))
|
||||
return yield* ProviderShared.unsupportedOperation({
|
||||
operation: "transcription.model",
|
||||
provider: PROVIDER,
|
||||
route: ADAPTER,
|
||||
message: `${request.model.id} is not a transcription model; use a transcribe model such as gemini-3.5-transcribe`,
|
||||
})
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
contents: [{ role: "user", parts: [yield* GeminiGenerateContent.mediaPart(ADAPTER, request.audio)] }],
|
||||
generationConfig: mergeJsonRecords(
|
||||
{
|
||||
audioTranscriptionConfig: {
|
||||
languageCodes: request.language === undefined ? undefined : [request.language],
|
||||
// Parts carry no offsets of their own, so segment times come from word offsets.
|
||||
wordTimestamp:
|
||||
request.diarize === true || request.timestamps === "word" || request.timestamps === "segment"
|
||||
? true
|
||||
: undefined,
|
||||
diarization: request.diarize === true ? true : undefined,
|
||||
},
|
||||
},
|
||||
request.providerOptions,
|
||||
),
|
||||
},
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const turn = (part: Schema.Schema.Type<typeof AudioTranscription>) => {
|
||||
const words = (part.words ?? []).map((word) => ({
|
||||
text: word.word,
|
||||
startSeconds: word.startOffset,
|
||||
endSeconds: word.endOffset,
|
||||
speaker: word.speakerLabel ?? part.speakerLabel,
|
||||
}))
|
||||
const first = words[0]
|
||||
const last = words.at(-1)
|
||||
return {
|
||||
text: part.text,
|
||||
words,
|
||||
segment:
|
||||
first === undefined || last === undefined
|
||||
? undefined
|
||||
: {
|
||||
text: part.text,
|
||||
startSeconds: first.startSeconds,
|
||||
endSeconds: last.endSeconds,
|
||||
speaker: part.speakerLabel,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const step = Effect.fn("GoogleTranscription.step")(function* (state: State, frame: string) {
|
||||
const chunk = yield* decodeChunk(frame)
|
||||
const blocked = GeminiGenerateContent.blocked(NAME, chunk, frame)
|
||||
if (blocked !== undefined) return yield* blocked
|
||||
const turns = (chunk.candidates?.[0]?.content?.parts ?? []).flatMap((part) =>
|
||||
part.audioTranscription === undefined ? [] : [turn(part.audioTranscription)],
|
||||
)
|
||||
const segments = turns.flatMap((item) => (item.segment === undefined ? [] : [item.segment]))
|
||||
state.words.push(...turns.flatMap((item) => item.words))
|
||||
state.segments.push(...segments)
|
||||
// Each part is one whole speaker turn without surrounding whitespace, so turns join with a space.
|
||||
const text = turns
|
||||
.map((item) => item.text)
|
||||
.filter((item) => item.length > 0)
|
||||
.join(" ")
|
||||
const delta = text.length === 0 || state.text.length === 0 ? text : ` ${text}`
|
||||
const events: ReadonlyArray<TranscriptionEvent> = [
|
||||
...(delta.length === 0 ? [] : [TranscriptionTextDeltaEvent.make({ delta })]),
|
||||
...segments.map((segment) => TranscriptionSegmentEvent.make({ segment })),
|
||||
]
|
||||
return [{ ...GeminiGenerateContent.track(state, chunk), text: state.text + delta }, events] as const
|
||||
})
|
||||
|
||||
const finish = (state: State) => {
|
||||
if (state.finishReason === undefined) return Effect.fail(MediaProtocol.incomplete(ADAPTER))
|
||||
return Effect.succeed([
|
||||
TranscriptionFinishEvent.make({
|
||||
text: state.text,
|
||||
segments: state.segments.length === 0 ? undefined : state.segments,
|
||||
words: state.words.length === 0 ? undefined : state.words,
|
||||
usage: GeminiGenerateContent.usage(state.usage),
|
||||
providerMetadata: GeminiGenerateContent.providerMetadata(state),
|
||||
}),
|
||||
])
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, TranscriptionEvent, string, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["prompt", "speakers"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes, context) => GeminiGenerateContent.frames(bytes, context.request.mode),
|
||||
initial: () => ({ text: "", segments: [], words: [] }),
|
||||
step,
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
TranscriptionModel.fromRoute<GoogleTranscriptionOptions, string, State>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => GeminiGenerateContent.path(request.model.id, request.mode),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const GoogleTranscription = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,233 @@
|
||||
import { Duration, Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import type { Status } from "../generation.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { VideoModel, VideoResponse, type VideoRequestFor } from "../video.js"
|
||||
import { ProviderShared, optionalArray } from "./shared.js"
|
||||
|
||||
const ADAPTER = "google-video"
|
||||
const NAME = "Google Veo"
|
||||
const PROVIDER = ProviderID.make("google")
|
||||
export const DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"
|
||||
/** Veo keeps generated files for two days; the asset carries that deadline so callers materialize in time. */
|
||||
const FILE_RETENTION = Duration.days(2)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type GoogleVideoString<Known extends string> = Known | (string & {})
|
||||
|
||||
/** Provider-native `parameters`. Common fields (`aspectRatio`, `resolution`, `durationSeconds`, `seed`) live on the request. */
|
||||
export type GoogleVideoOptions = {
|
||||
readonly personGeneration?: GoogleVideoString<"allow_all" | "allow_adult" | "dont_allow">
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = VideoRequestFor<GoogleVideoOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Token and response schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** The long-running operation name, e.g. `models/veo-3.1-generate-preview/operations/abc123`. */
|
||||
export const Token = Schema.Struct({ operation: Schema.String })
|
||||
export type Token = Schema.Schema.Type<typeof Token>
|
||||
|
||||
const StartResponse = Schema.Struct({ name: Schema.String })
|
||||
|
||||
const Operation = Schema.Struct({
|
||||
done: Schema.optional(Schema.Boolean),
|
||||
error: Schema.optional(Schema.Struct({ message: Schema.optional(Schema.String) })),
|
||||
response: Schema.optional(
|
||||
Schema.Struct({
|
||||
generateVideoResponse: Schema.optional(
|
||||
Schema.Struct({
|
||||
generatedSamples: optionalArray(
|
||||
Schema.Struct({
|
||||
video: Schema.optional(
|
||||
Schema.Struct({
|
||||
uri: Schema.optional(Schema.String),
|
||||
mimeType: Schema.optional(Schema.String),
|
||||
}),
|
||||
),
|
||||
}),
|
||||
),
|
||||
raiMediaFilteredCount: Schema.optional(Schema.Number),
|
||||
raiMediaFilteredReasons: optionalArray(Schema.String),
|
||||
}),
|
||||
),
|
||||
}),
|
||||
),
|
||||
metadata: Schema.optional(Schema.Unknown),
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Veo takes inline media only; a prior Veo output is `Media.url` with transient auth, so materialize it first.
|
||||
const inlineMedia = (asset: Media.Asset) =>
|
||||
ProviderShared.requireInlineMedia(NAME, asset).pipe(
|
||||
Effect.map((inline) => ({ inlineData: { mimeType: inline.mime, data: inline.base64 } })),
|
||||
)
|
||||
|
||||
const fromRequest = Effect.fn("GoogleVideo.fromRequest")(function* (request: Request) {
|
||||
if (request.n !== undefined && request.n > 1)
|
||||
return yield* ProviderShared.unsupportedOperation({
|
||||
operation: "video.n",
|
||||
provider: PROVIDER,
|
||||
route: ADAPTER,
|
||||
message: `${NAME} generates one video per request; call it once per video instead of n=${request.n}`,
|
||||
})
|
||||
if (request.audio === false)
|
||||
return yield* ProviderShared.unsupportedOperation({
|
||||
operation: "video.audio",
|
||||
provider: PROVIDER,
|
||||
route: ADAPTER,
|
||||
message: `${NAME} always generates audio; audio: false cannot be honored`,
|
||||
})
|
||||
if (request.frames?.last !== undefined && request.frames.first === undefined)
|
||||
return yield* ProviderShared.invalidRequest(`${NAME} requires frames.first when frames.last is set`)
|
||||
const image = request.frames?.first === undefined ? undefined : yield* inlineMedia(request.frames.first)
|
||||
const lastFrame = request.frames?.last === undefined ? undefined : yield* inlineMedia(request.frames.last)
|
||||
const video = request.video === undefined ? undefined : yield* inlineMedia(request.video)
|
||||
const referenceImages = yield* Effect.forEach(request.references ?? [], (asset) =>
|
||||
inlineMedia(asset).pipe(Effect.map((image) => ({ image, referenceType: "asset" }))),
|
||||
)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
instances: [
|
||||
{
|
||||
prompt: request.prompt,
|
||||
image,
|
||||
lastFrame,
|
||||
referenceImages: referenceImages.length === 0 ? undefined : referenceImages,
|
||||
video,
|
||||
},
|
||||
],
|
||||
parameters: mergeJsonRecords(
|
||||
{
|
||||
aspectRatio: request.aspectRatio,
|
||||
resolution: request.resolution,
|
||||
durationSeconds: request.durationSeconds,
|
||||
negativePrompt: request.negativePrompt,
|
||||
seed: request.seed,
|
||||
},
|
||||
request.providerOptions,
|
||||
),
|
||||
},
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeStart = MediaProtocol.decodeStarted(ADAPTER, NAME, StartResponse, (value) => ({
|
||||
token: { operation: value.name },
|
||||
snapshot: { id: value.name, status: "running" },
|
||||
}))
|
||||
|
||||
// Operations carry no status string: not done is running, done with `error` failed, otherwise completed.
|
||||
const statusOf = (operation: typeof Operation.Type): Status => {
|
||||
if (operation.done !== true) return "running"
|
||||
return operation.error === undefined ? "completed" : "failed"
|
||||
}
|
||||
|
||||
const decodeOperation = MediaProtocol.decodeJson(ADAPTER, NAME, Operation)
|
||||
|
||||
const decodeStatus = Effect.fn("GoogleVideo.decodeStatus")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeOperation(response)
|
||||
return { id: context.token.operation, status: statusOf(output.value) }
|
||||
})
|
||||
|
||||
const decodeResult = Effect.fn("GoogleVideo.decodeResult")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeOperation(response)
|
||||
const operation = output.value
|
||||
const status = statusOf(operation)
|
||||
if (status === "running")
|
||||
return yield* output.invalid(`${NAME} operation ${context.token.operation} has not finished`)
|
||||
if (status === "failed")
|
||||
return yield* output.ended(
|
||||
"failed",
|
||||
`${NAME} operation failed${operation.error?.message === undefined ? "" : `: ${operation.error.message}`}`,
|
||||
)
|
||||
const generated = operation.response?.generateVideoResponse
|
||||
// Downloads require the same API key as the poll; the asset carries it transiently and follows the redirect.
|
||||
const videos = yield* Effect.forEach(
|
||||
(generated?.generatedSamples ?? []).flatMap((sample) =>
|
||||
sample.video?.uri === undefined ? [] : [{ uri: sample.video.uri, mimeType: sample.video.mimeType }],
|
||||
),
|
||||
(video) =>
|
||||
MediaProtocol.expiringUrl(video.uri, FILE_RETENTION, {
|
||||
mediaType: video.mimeType ?? "video/mp4",
|
||||
headers: context.auth,
|
||||
}),
|
||||
)
|
||||
const reasons = generated?.raiMediaFilteredReasons ?? []
|
||||
const notices = reasons.map((reason) => ({
|
||||
type: "filtered" as const,
|
||||
message: `${NAME} filtered media: ${reason}`,
|
||||
providerMetadata: { google: { raiMediaFilteredReason: reason } },
|
||||
}))
|
||||
if (videos.length === 0 && (reasons.length > 0 || (generated?.raiMediaFilteredCount ?? 0) > 0))
|
||||
return yield* output.contentPolicy(
|
||||
`${NAME} filtered every video${reasons.length === 0 ? "" : `: ${reasons.join("; ")}`}`,
|
||||
)
|
||||
if (videos.length === 0) return yield* output.invalid(`${NAME} operation completed without any video`)
|
||||
return new VideoResponse({
|
||||
videos,
|
||||
notices: notices.length === 0 ? undefined : notices,
|
||||
providerMetadata: {
|
||||
google: {
|
||||
operation: context.token.operation,
|
||||
raiMediaFilteredCount: generated?.raiMediaFilteredCount,
|
||||
metadata: operation.metadata,
|
||||
},
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const operationPath = (token: Token) => `/${token.operation}`
|
||||
|
||||
export const protocol = MediaProtocol.queued<Request, VideoResponse, Token>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
token: Token,
|
||||
start: { body: { from: fromRequest }, decode: decodeStart },
|
||||
status: { path: operationPath, decode: decodeStatus },
|
||||
result: { path: operationPath, decode: decodeResult },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
VideoModel.fromRoute<GoogleVideoOptions, Token>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => `/models/${request.model.id}:predictLongRunning`,
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const GoogleVideo = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -1,17 +1,25 @@
|
||||
import { Effect, Encoding, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import { GeneratedImage, ImageModel, ImageResponse, type ImageRequestFor, type ImageRoute } from "../image.js"
|
||||
import { Auth } from "../route/auth.js"
|
||||
import { Usage, mergeHttpOptions, mergeJsonRecords, type HttpOptions } from "../schema/index.js"
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { JsonObject, ProviderShared, optionalNull } from "./shared.js"
|
||||
import { ImageInputs } from "./utils/image-input.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "meta-images"
|
||||
const NAME = "Meta Images"
|
||||
const PROVIDER = ProviderID.make("meta")
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type OpenString<Known extends string> = Known | (string & {})
|
||||
|
||||
/** Provider-native options. Common fields (`n`, `size`, `format`, `images`) live on the request. */
|
||||
export type ImageOptions = {
|
||||
readonly n?: number
|
||||
/** Aspect ratio hint, not an exact output resolution. */
|
||||
readonly size?: string
|
||||
readonly outputFormat?: OpenString<"webp" | "png" | "jpeg">
|
||||
readonly responseFormat?: OpenString<"b64_json" | "url">
|
||||
readonly reasoningStrength?: OpenString<"low" | "high">
|
||||
readonly toolEnablement?: {
|
||||
@@ -22,12 +30,19 @@ export type ImageOptions = {
|
||||
readonly [key: string]: unknown
|
||||
}
|
||||
|
||||
export type Request = ImageRequestFor<ImageOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Request body and response schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const Body = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
model: Schema.String,
|
||||
prompt: Schema.String,
|
||||
images: Schema.optional(Schema.Array(JsonObject)),
|
||||
n: Schema.optional(Schema.Number),
|
||||
/** Aspect ratio hint, not an exact output resolution. */
|
||||
size: Schema.optional(Schema.String),
|
||||
output_format: Schema.optional(Schema.String),
|
||||
response_format: Schema.optional(Schema.String),
|
||||
@@ -49,85 +64,98 @@ const Response = Schema.Struct({
|
||||
),
|
||||
})
|
||||
|
||||
export const model = (input: {
|
||||
readonly id: string
|
||||
readonly auth: Auth.Definition
|
||||
readonly baseURL: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}) => {
|
||||
const route: ImageRoute<ImageOptions> = {
|
||||
id: "meta-images",
|
||||
generate: Effect.fn("MetaImages.generate")(function* (request: ImageRequestFor<ImageOptions>, execute) {
|
||||
const http = mergeHttpOptions(request.model.http, request.http)
|
||||
const images = yield* Effect.forEach(request.images ?? [], (image) => {
|
||||
if (image.type === "bytes") return Effect.succeed({ image_url: ImageInputs.dataUrl(image) })
|
||||
if (image.type === "url") return Effect.succeed({ image_url: image.url })
|
||||
return ImageInputs.invalid("Meta Images accepts image bytes and URLs")
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const isEdit = (request: Request) => (request.images?.length ?? 0) > 0
|
||||
|
||||
// Meta has no file handles: refs are rejected even when they name this provider.
|
||||
const reference = (asset: Media.Asset) =>
|
||||
ProviderShared.mediaReference(asset, undefined, NAME).pipe(Effect.map((item) => ({ image_url: item.value })))
|
||||
|
||||
const fromRequest = Effect.fn("MetaImages.fromRequest")(function* (request: Request) {
|
||||
const images = yield* Effect.forEach(request.images ?? [], reference)
|
||||
const { responseFormat, reasoningStrength, toolEnablement, ...native } = request.providerOptions ?? {}
|
||||
const payload = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Body))(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
images: images.length === 0 ? undefined : images,
|
||||
n: request.n,
|
||||
size: request.size,
|
||||
output_format: request.format,
|
||||
response_format: responseFormat,
|
||||
reasoning_strength: reasoningStrength,
|
||||
tool_enablement: toolEnablement,
|
||||
},
|
||||
native,
|
||||
request.http?.body,
|
||||
),
|
||||
)
|
||||
return MediaProtocol.json(payload)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeResponse = Effect.fn("MetaImages.decodeResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.DecodeContext<Request>,
|
||||
) {
|
||||
const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, Response)(response)
|
||||
const decoded = output.value
|
||||
const requested = context.body.type === "json" ? context.body.value.output_format : undefined
|
||||
const format = decoded.output_format ?? (typeof requested === "string" ? requested : "webp")
|
||||
const mediaType = `image/${format}`
|
||||
const images = yield* Effect.forEach(decoded.data, (item, index) => {
|
||||
if (item.b64_json)
|
||||
return MediaInput.decodedAsset(output.invalid, `${NAME} result ${index}`, item.b64_json, mediaType, {
|
||||
info: { format },
|
||||
})
|
||||
const { outputFormat, responseFormat, reasoningStrength, toolEnablement, ...native } = request.options ?? {}
|
||||
const payload = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(Body))(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
images: images.length === 0 ? undefined : images,
|
||||
output_format: outputFormat,
|
||||
response_format: responseFormat,
|
||||
reasoning_strength: reasoningStrength,
|
||||
tool_enablement: toolEnablement,
|
||||
if (item.url) return Effect.succeed(Media.url(item.url, { mediaType, info: { format } }))
|
||||
return Effect.fail(output.invalid(`${NAME} result ${index} has neither image data nor a URL`))
|
||||
})
|
||||
if (images.length === 0) return yield* output.invalid(`${NAME} returned no images`)
|
||||
return new ImageResponse({
|
||||
images,
|
||||
usage:
|
||||
decoded.usage === undefined
|
||||
? undefined
|
||||
: {
|
||||
type: "tokens",
|
||||
input: decoded.usage.input_tokens,
|
||||
output: decoded.usage.output_tokens,
|
||||
total: decoded.usage.total_tokens,
|
||||
details: { meta: decoded.usage },
|
||||
},
|
||||
native,
|
||||
http?.body,
|
||||
),
|
||||
)
|
||||
const body = ProviderShared.encodeJson(payload)
|
||||
const url = new URL(`${input.baseURL.replace(/\/$/, "")}/images/${images.length === 0 ? "generations" : "edits"}`)
|
||||
Object.entries(http?.query ?? {}).forEach(([key, value]) => url.searchParams.set(key, value))
|
||||
const headers = yield* Auth.toEffect(input.auth)({
|
||||
request,
|
||||
method: "POST",
|
||||
url: url.toString(),
|
||||
body,
|
||||
headers: Headers.fromInput({ ...input.headers, ...http?.headers }),
|
||||
})
|
||||
const response = yield* execute(
|
||||
HttpClientRequest.post(url.toString()).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(body, "application/json"),
|
||||
),
|
||||
)
|
||||
const output = yield* ProviderShared.imageResponse("meta-images", "Meta Images", response)
|
||||
const decoded = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(Response))(output.body).pipe(
|
||||
Effect.mapError((cause) => output.invalid("Meta Images returned an invalid response", cause)),
|
||||
)
|
||||
const format = decoded.output_format ?? payload.output_format ?? "webp"
|
||||
const generated = yield* Effect.forEach(decoded.data, (item, index) => {
|
||||
if (item.b64_json)
|
||||
return Effect.fromResult(Encoding.decodeBase64(item.b64_json)).pipe(
|
||||
Effect.mapError((cause) => output.invalid(`Meta Images result ${index} contains invalid base64`, cause)),
|
||||
Effect.map((data) => new GeneratedImage({ mediaType: `image/${format}`, data })),
|
||||
)
|
||||
if (item.url) return Effect.succeed(new GeneratedImage({ mediaType: `image/${format}`, data: item.url }))
|
||||
return output.invalid(`Meta Images result ${index} has neither image data nor a URL`)
|
||||
})
|
||||
if (generated.length === 0) return yield* output.invalid("Meta Images returned no images")
|
||||
return new ImageResponse({
|
||||
images: generated,
|
||||
usage:
|
||||
decoded.usage === undefined
|
||||
? undefined
|
||||
: new Usage({
|
||||
inputTokens: decoded.usage.input_tokens,
|
||||
outputTokens: decoded.usage.output_tokens,
|
||||
totalTokens: decoded.usage.total_tokens,
|
||||
providerMetadata: { meta: decoded.usage },
|
||||
}),
|
||||
providerMetadata: { meta: { outputFormat: format } },
|
||||
})
|
||||
}),
|
||||
}
|
||||
return ImageModel.make<ImageOptions>({ id: input.id, provider: "meta", route, http: input.http })
|
||||
}
|
||||
providerMetadata: { meta: { outputFormat: format } },
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.inline<Request, ImageResponse>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["mask", "aspectRatio", "seed"],
|
||||
body: { from: fromRequest },
|
||||
response: { decode: decodeResponse },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput & { readonly baseURL: string }) =>
|
||||
ImageModel.fromRoute<ImageOptions>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
path: ({ request }) => `/images/${isEdit(request) ? "edits" : "generations"}`,
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export * as MetaImages from "./meta-images.js"
|
||||
|
||||
@@ -224,11 +224,13 @@ type MistralEvent = Schema.Schema.Type<typeof MistralEvent>
|
||||
const MistralStreamEvent = Schema.Union([Schema.Literal(DONE), Protocol.jsonEvent(MistralEvent)])
|
||||
|
||||
const lowerMedia = Effect.fn("MistralChat.lowerMedia")(function* (part: MediaPart) {
|
||||
const media = ProviderShared.normalizeMedia(part)
|
||||
const url = typeof part.data === "string" && /^(?:https?:|data:)/.test(part.data) ? part.data : media.dataUrl
|
||||
if (media.mime.startsWith("image/")) return { type: "image_url" as const, image_url: url }
|
||||
if (media.mime === "application/pdf") return { type: "document_url" as const, document_url: url }
|
||||
return yield* ProviderShared.invalidRequest(`Mistral Chat does not support media type ${part.mediaType}`)
|
||||
const mime = part.media.mediaType.toLowerCase()
|
||||
const url =
|
||||
ProviderShared.mediaUrl(part.media) ??
|
||||
(yield* ProviderShared.requireInlineMedia("Mistral Chat", part.media)).dataUrl
|
||||
if (mime.startsWith("image/")) return { type: "image_url" as const, image_url: url }
|
||||
if (mime === "application/pdf") return { type: "document_url" as const, document_url: url }
|
||||
return yield* ProviderShared.invalidRequest(`Mistral Chat does not support media type ${part.media.mediaType}`)
|
||||
})
|
||||
|
||||
const lowerUser = Effect.fn("MistralChat.lowerUser")(function* (message: LLMRequest["messages"][number]) {
|
||||
@@ -316,7 +318,7 @@ const lowerToolResults = Effect.fn("MistralChat.lowerToolResults")(function* (
|
||||
content.push({ type: "text", text: item.text })
|
||||
continue
|
||||
}
|
||||
content.push(yield* lowerMedia({ type: "media", mediaType: item.mime, data: item.uri, filename: item.name }))
|
||||
content.push(yield* lowerMedia(ProviderShared.toolFileMedia(item)))
|
||||
}
|
||||
output.push({
|
||||
role: "tool",
|
||||
|
||||
@@ -73,11 +73,6 @@ const driver = (options: Options, body: string): WebSocketChannelDriver => {
|
||||
)
|
||||
if (event.type === "error") {
|
||||
terminal = true
|
||||
yield* OpenResponses.decodeKnownErrorEvent(event).pipe(
|
||||
Effect.mapError((cause) =>
|
||||
ProviderShared.eventError(options.id, `${options.name} returned a malformed error event`, frame, cause),
|
||||
),
|
||||
)
|
||||
return {
|
||||
type: "provider-failure",
|
||||
error: OpenResponses.providerFailure(event, `${options.name} stream error`, frame),
|
||||
|
||||
@@ -108,7 +108,7 @@ const incremental = (
|
||||
return input.slice(baseline.length)
|
||||
}
|
||||
|
||||
const code = (event: OpenResponses.Event) => event.code || event.error?.code || event.response?.error?.code || undefined
|
||||
const code = (event: OpenResponses.Event) => OpenResponses.errorDetail(event).code
|
||||
|
||||
const rejected = (
|
||||
observation: Extract<ChannelObservation, { readonly type: "provider-failure" }>,
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { Effect, Option, Schema, SchemaGetter } from "effect"
|
||||
import { Effect, Option, Schema } from "effect"
|
||||
import type { Content } from "@opencode/schema/tool"
|
||||
import { HttpTransport } from "../route/transport/index.js"
|
||||
import { Protocol } from "../route/protocol.js"
|
||||
@@ -18,6 +18,7 @@ import {
|
||||
type ToolDefinition,
|
||||
type ToolResultPart,
|
||||
} from "../schema/index.js"
|
||||
import type { Media } from "../media.js"
|
||||
import { JsonObject, optionalArray, optionalNull, ProviderShared } from "./shared.js"
|
||||
import { classifyProviderFailure } from "../provider-error.js"
|
||||
import { effortUpdate } from "../effort-updates.js"
|
||||
@@ -333,53 +334,13 @@ export const StreamItem = Schema.StructWithRest(
|
||||
export type StreamItem = Schema.Schema.Type<typeof StreamItem>
|
||||
export type OutputItem = StreamItem & { readonly id: string }
|
||||
|
||||
// Responses-compatible providers put streaming error details at the top level or
|
||||
// under `error`, and response failures under `response.error`. Accept all three shapes.
|
||||
// Responses-compatible providers put error details at the top level, under `error`, or under
|
||||
// `response.error`, and gateways reshape them freely: strings, numeric codes, extra fields. Those
|
||||
// fields decode as opaque values and `errorDetail` reads them defensively, so an error frame can
|
||||
// only fail on invalid JSON and otherwise always classifies with the raw body as the fallback.
|
||||
// https://www.openresponses.org/specification
|
||||
const OpenResponsesErrorObject = Schema.Struct({
|
||||
type: optionalNull(Schema.String),
|
||||
code: optionalNull(Schema.String),
|
||||
message: optionalNull(Schema.String),
|
||||
param: optionalNull(Schema.String),
|
||||
})
|
||||
const OpenResponsesErrorPayload = Schema.Union([Schema.String, OpenResponsesErrorObject]).pipe(
|
||||
Schema.decodeTo(OpenResponsesErrorObject, {
|
||||
decode: SchemaGetter.transform((error) => (typeof error === "string" ? { message: error } : error)),
|
||||
encode: SchemaGetter.passthrough(),
|
||||
}),
|
||||
)
|
||||
type OpenResponsesErrorPayload = Schema.Schema.Type<typeof OpenResponsesErrorPayload>
|
||||
|
||||
const WebSocketErrorHeader = Schema.Union([Schema.String, Schema.Number, Schema.Boolean])
|
||||
export const WebSocketErrorEvent = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
type: Schema.tag("error"),
|
||||
status: Schema.optional(Schema.Number),
|
||||
status_code: Schema.optional(Schema.Number),
|
||||
code: optionalNull(Schema.String),
|
||||
message: Schema.optional(Schema.String),
|
||||
param: optionalNull(Schema.String),
|
||||
error: optionalNull(OpenResponsesErrorPayload),
|
||||
headers: Schema.optional(Schema.Record(Schema.String, WebSocketErrorHeader)),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
)
|
||||
const decodeWebSocketErrorEvent = Schema.decodeUnknownEffect(WebSocketErrorEvent)
|
||||
|
||||
export const decodeKnownErrorEvent = (event: Event) =>
|
||||
decodeWebSocketErrorEvent({
|
||||
...event,
|
||||
status: typeof event.status === "number" ? event.status : undefined,
|
||||
status_code: typeof event.status_code === "number" ? event.status_code : undefined,
|
||||
headers: ProviderShared.isRecord(event.headers)
|
||||
? Object.fromEntries(
|
||||
Object.entries(event.headers).filter(
|
||||
(entry): entry is [string, string | number | boolean] =>
|
||||
typeof entry[1] === "string" || typeof entry[1] === "number" || typeof entry[1] === "boolean",
|
||||
),
|
||||
)
|
||||
: undefined,
|
||||
})
|
||||
const asText = (value: unknown) =>
|
||||
typeof value === "string" && value.length > 0 ? value : typeof value === "number" ? String(value) : undefined
|
||||
|
||||
export const Event = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
@@ -400,31 +361,18 @@ export const Event = Schema.StructWithRest(
|
||||
incomplete_details: optionalNull(Schema.Struct({ reason: Schema.optional(Schema.String) })),
|
||||
output: Schema.optional(Schema.Array(StreamItem)),
|
||||
usage: optionalNull(OpenResponsesUsage),
|
||||
error: optionalNull(OpenResponsesErrorPayload),
|
||||
error: Schema.optional(Schema.Unknown),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
),
|
||||
),
|
||||
code: optionalNull(Schema.String),
|
||||
message: Schema.optional(Schema.String),
|
||||
param: optionalNull(Schema.String),
|
||||
error: optionalNull(OpenResponsesErrorPayload),
|
||||
code: Schema.optional(Schema.Unknown),
|
||||
message: Schema.optional(Schema.Unknown),
|
||||
error: Schema.optional(Schema.Unknown),
|
||||
status: Schema.optional(Schema.Unknown),
|
||||
status_code: Schema.optional(Schema.Unknown),
|
||||
headers: Schema.optional(Schema.Unknown),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
).pipe(
|
||||
Schema.decode({
|
||||
decode: SchemaGetter.transform((event) => {
|
||||
if (event.type !== "error" || event.error != null) return event
|
||||
const { code, message, param, ...rest } = event
|
||||
if (code === undefined && message === undefined && param === undefined) return event
|
||||
// Flat errors (for example, Meta's) can also arrive through generic Responses endpoints.
|
||||
return { ...rest, error: { code, message, param } }
|
||||
}),
|
||||
encode: SchemaGetter.passthrough(),
|
||||
}),
|
||||
)
|
||||
export type Event = Schema.Schema.Type<typeof Event>
|
||||
export type NormalizedEvent = Event & { readonly item?: OutputItem | null }
|
||||
@@ -433,16 +381,15 @@ const decodeEventValue = Schema.decodeUnknownEffect(Event)
|
||||
const decodeFrame = Schema.decodeUnknownEffect(ProviderShared.Json)
|
||||
|
||||
/**
|
||||
* Decodes one WebSocket frame. xAI answers a rejected `response.create` with `{ "error": { "message", "type" } }` and no
|
||||
* event type; that envelope reads as an error event so the failure classifies instead of failing decoding.
|
||||
* Decodes one WebSocket frame. Some providers and gateways answer a rejected `response.create` with a bare
|
||||
* `{ "error": ... }` envelope and no event type; that reads as an error event so it classifies instead of
|
||||
* failing decoding.
|
||||
*/
|
||||
export const decodeChannelEvent = (frame: string) =>
|
||||
decodeFrame(frame).pipe(
|
||||
Effect.flatMap((value) =>
|
||||
decodeEventValue(
|
||||
ProviderShared.isRecord(value) &&
|
||||
value.type === undefined &&
|
||||
(typeof value.error === "string" || ProviderShared.isRecord(value.error))
|
||||
ProviderShared.isRecord(value) && value.type === undefined && value.error != null
|
||||
? { ...value, type: "error" }
|
||||
: value,
|
||||
),
|
||||
@@ -457,7 +404,7 @@ export interface ProviderAdapter {
|
||||
) => Effect.Effect<{ readonly type: string }, AIError>
|
||||
readonly lowerMedia?: (input: {
|
||||
readonly part: MediaPart
|
||||
readonly media: ProviderShared.NormalizedMedia
|
||||
readonly media: Media.Inline | undefined
|
||||
readonly request: LLMRequest
|
||||
}) => MediaInput | undefined
|
||||
readonly restoreHostedToolItem?: (item: unknown) => HostedToolReplayItem | undefined
|
||||
@@ -564,29 +511,28 @@ const lowerMedia = Effect.fn("OpenResponses.lowerMedia")(function* (
|
||||
adapter: ProviderAdapter,
|
||||
target: "message" | "tool-result",
|
||||
) {
|
||||
const media = ProviderShared.normalizeMedia(part)
|
||||
const media = part.media.inline()
|
||||
const providerMedia = adapter.lowerMedia?.({ part, media, request })
|
||||
if (providerMedia) return providerMedia
|
||||
const detail = yield* ProviderShared.validateWith(Schema.decodeUnknownEffect(OpenResponsesInputImage.fields.detail))(
|
||||
part.providerMetadata?.[metadataKey(request.model)]?.detail,
|
||||
)
|
||||
const url =
|
||||
typeof part.data === "string" && (part.data.startsWith("https://") || part.data.startsWith("http://"))
|
||||
? part.data
|
||||
: undefined
|
||||
if (!media.mime.startsWith("image/")) {
|
||||
if (target === "tool-result" && media.mime.startsWith("video/"))
|
||||
return { type: "input_video" as const, video_url: url ?? media.dataUrl }
|
||||
const mime = part.media.mediaType.toLowerCase()
|
||||
const url = ProviderShared.mediaUrl(part.media)
|
||||
const location = url ?? (yield* ProviderShared.requireInlineMedia(adapter.name, part.media)).dataUrl
|
||||
if (part.media.kind !== "image") {
|
||||
if (target === "tool-result" && part.media.kind === "video")
|
||||
return { type: "input_video" as const, video_url: location }
|
||||
return {
|
||||
type: "input_file" as const,
|
||||
filename: part.filename ?? (media.mime === "application/pdf" ? "document.pdf" : "file"),
|
||||
filename: part.filename ?? (mime === "application/pdf" ? "document.pdf" : "file"),
|
||||
detail,
|
||||
...(url ? { file_url: url } : { file_data: media.dataUrl }),
|
||||
...(url ? { file_url: url } : { file_data: location }),
|
||||
}
|
||||
}
|
||||
return {
|
||||
type: "input_image" as const,
|
||||
image_url: url ?? media.dataUrl,
|
||||
image_url: location,
|
||||
detail,
|
||||
}
|
||||
})
|
||||
@@ -616,12 +562,7 @@ const lowerToolResultContentItem = Effect.fnUntraced(function* (
|
||||
adapter: ProviderAdapter,
|
||||
) {
|
||||
if (item.type === "text") return { type: "input_text" as const, text: item.text }
|
||||
return yield* lowerMedia(
|
||||
{ type: "media", mediaType: item.mime, data: item.uri, filename: item.name },
|
||||
request,
|
||||
adapter,
|
||||
"tool-result",
|
||||
)
|
||||
return yield* lowerMedia(ProviderShared.toolFileMedia(item), request, adapter, "tool-result")
|
||||
})
|
||||
|
||||
const lowerHostedToolResultContentItem = Effect.fnUntraced(function* (
|
||||
@@ -630,11 +571,7 @@ const lowerHostedToolResultContentItem = Effect.fnUntraced(function* (
|
||||
adapter: ProviderAdapter,
|
||||
) {
|
||||
if (item.type === "text") return { type: "input_text" as const, text: item.text }
|
||||
return yield* lowerMessageMedia(
|
||||
{ type: "media", mediaType: item.mime, data: item.uri, filename: item.name },
|
||||
request,
|
||||
adapter,
|
||||
)
|
||||
return yield* lowerMessageMedia(ProviderShared.toolFileMedia(item), request, adapter)
|
||||
})
|
||||
|
||||
const lowerToolResultOutput = Effect.fnUntraced(function* (
|
||||
@@ -780,11 +717,22 @@ const lowerMessages = Effect.fn("OpenResponses.lowerMessages")(function* (
|
||||
})
|
||||
continue
|
||||
}
|
||||
if (part.type === "media") {
|
||||
flushText()
|
||||
// Responses has no assistant-authored image item; replay generated media (e.g. from Gemini) as user input.
|
||||
input.push({
|
||||
type: "message",
|
||||
role: "user",
|
||||
content: [yield* lowerMessageMedia(part, request, adapter)],
|
||||
})
|
||||
continue
|
||||
}
|
||||
return yield* ProviderShared.unsupportedContent(adapter.name, "assistant", [
|
||||
"text",
|
||||
"reasoning",
|
||||
"tool-call",
|
||||
"tool-result",
|
||||
"media",
|
||||
])
|
||||
}
|
||||
flushText()
|
||||
@@ -1422,22 +1370,21 @@ const onResponseFinish = Effect.fn("OpenResponses.onResponseFinish")(function* (
|
||||
return [{ ...current, lifecycle }, events] satisfies StepResult
|
||||
})
|
||||
|
||||
// Build the prettiest summary available from whatever the provider supplied.
|
||||
// When both code and message are present, prefix the code so consumers see
|
||||
// the failure mode (e.g. `rate_limit_exceeded: Slow down`) instead of just
|
||||
// the bare message — production rate limits and context-length failures used
|
||||
// to be indistinguishable from generic stream drops. Returns undefined when
|
||||
// the payload carries no usable summary.
|
||||
const providerErrorMessage = (event: Event, nested: OpenResponsesErrorPayload | undefined): string | undefined => {
|
||||
const message = event.message || nested?.message || undefined
|
||||
const code = event.code || nested?.code || undefined
|
||||
if (message && code) return `${code}: ${message}`
|
||||
return message || code
|
||||
/** Error code and message from wherever the frame put them; top-level fields win over nested ones. */
|
||||
export const errorDetail = (event: Event) => {
|
||||
const raw = event.error ?? event.response?.error
|
||||
const nested = typeof raw === "string" ? { message: raw } : ProviderShared.isRecord(raw) ? raw : undefined
|
||||
return {
|
||||
message: asText(event.message) ?? asText(nested?.message),
|
||||
code: asText(event.code) ?? asText(nested?.code),
|
||||
}
|
||||
}
|
||||
|
||||
// Prefix the code when both are present (`rate_limit_exceeded: Slow down`) so the failure mode is
|
||||
// visible; fall back to the raw frame rather than a generic message when neither decodes.
|
||||
export const providerFailure = (event: Event, fallback: string, body = ProviderShared.encodeJson(event)) => {
|
||||
const nested = event.error ?? event.response?.error ?? undefined
|
||||
const summary = providerErrorMessage(event, nested)
|
||||
const detail = errorDetail(event)
|
||||
const summary = detail.message && detail.code ? `${detail.code}: ${detail.message}` : (detail.message ?? detail.code)
|
||||
const message = summary ?? (body === "{}" ? fallback : body)
|
||||
const status =
|
||||
typeof event.status === "number"
|
||||
@@ -1520,18 +1467,7 @@ export const step = (state: ParserState, event: NormalizedEvent) => {
|
||||
if (event.type === "response.output_item.done") return onOutputItemDone(state, event.item)
|
||||
if (event.type === "response.completed" || event.type === "response.incomplete") return onResponseFinish(state, event)
|
||||
if (event.type === "response.failed") return providerFailure(event, `${state.name} response failed`)
|
||||
if (event.type === "error")
|
||||
return decodeKnownErrorEvent(event).pipe(
|
||||
Effect.mapError((cause) =>
|
||||
ProviderShared.eventError(
|
||||
state.id,
|
||||
`${state.name} returned a malformed error event`,
|
||||
ProviderShared.encodeJson(event),
|
||||
cause,
|
||||
),
|
||||
),
|
||||
Effect.flatMap(() => providerFailure(event, `${state.name} stream error`)),
|
||||
)
|
||||
if (event.type === "error") return providerFailure(event, `${state.name} stream error`)
|
||||
return Effect.succeed<StepResult>([state, NO_EVENTS])
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { Effect, Option, Schema } from "effect"
|
||||
import { Tool } from "@opencode/schema/tool"
|
||||
import { Route } from "../route/client.js"
|
||||
import { Auth } from "../route/auth.js"
|
||||
@@ -76,6 +76,44 @@ const OpenAIChatAssistantToolCall = Schema.Struct({
|
||||
})
|
||||
type OpenAIChatAssistantToolCall = Schema.Schema.Type<typeof OpenAIChatAssistantToolCall>
|
||||
|
||||
// `reasoning_details` carries two dialects. OpenRouter's `reasoning.*` entries
|
||||
// must be replayed unmodified (`index` included), so they keep every field they
|
||||
// arrived with. Kimi's OpenAI-compatible surface streams preserved thinking as
|
||||
// bare `summary` / `encrypted` entries keyed by a stream-only `index`; Kimi does
|
||||
// not document this publicly, so the handling follows Kimi Code (Kimi's own
|
||||
// client): merge summary deltas by `index`, replay without `index`, and always
|
||||
// send `reasoning_content` alongside. Anything else is dropped at the boundary.
|
||||
const OpenRouterDetailFields = {
|
||||
id: Schema.optional(Schema.NullOr(Schema.String)),
|
||||
format: Schema.optional(Schema.String),
|
||||
index: Schema.optional(Schema.Number),
|
||||
signature: Schema.optional(Schema.NullOr(Schema.String)),
|
||||
}
|
||||
const ReasoningDetail = Schema.Union([
|
||||
Schema.StructWithRest(
|
||||
Schema.Struct({ type: Schema.Literal("reasoning.text"), text: Schema.optional(Schema.String), ...OpenRouterDetailFields }),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
),
|
||||
Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
type: Schema.Literal("reasoning.summary"),
|
||||
summary: Schema.optional(Schema.String),
|
||||
...OpenRouterDetailFields,
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
),
|
||||
Schema.StructWithRest(
|
||||
Schema.Struct({ type: Schema.Literal("reasoning.encrypted"), data: Schema.String, ...OpenRouterDetailFields }),
|
||||
[Schema.Record(Schema.String, Schema.Unknown)],
|
||||
),
|
||||
Schema.Struct({ type: Schema.Literal("summary"), summary: Schema.String, index: Schema.optional(Schema.Number) }),
|
||||
Schema.Struct({ type: Schema.Literal("encrypted"), encrypted: Schema.String, index: Schema.optional(Schema.Number) }),
|
||||
])
|
||||
type ReasoningDetail = Schema.Schema.Type<typeof ReasoningDetail>
|
||||
const decodeReasoningDetail = Schema.decodeUnknownOption(ReasoningDetail)
|
||||
const knownReasoningDetails = (details: ReadonlyArray<unknown>) =>
|
||||
details.flatMap((detail) => Option.toArray(decodeReasoningDetail(detail)))
|
||||
|
||||
// Intentionally omit Gemini's provider-specific `extra_content.google.thought_signature`
|
||||
// extension until direct Google OpenAI-compatible routing is supported here:
|
||||
// https://github.com/vercel/ai/issues/11590
|
||||
@@ -92,6 +130,10 @@ const OpenAIChatUserContent = Schema.Union([
|
||||
type: Schema.Literal("image_url"),
|
||||
image_url: Schema.Struct({ url: Schema.String }),
|
||||
}),
|
||||
Schema.Struct({
|
||||
type: Schema.Literal("file"),
|
||||
file: Schema.Struct({ filename: Schema.String, file_data: Schema.String }),
|
||||
}),
|
||||
])
|
||||
|
||||
const OpenAIChatMessage = Schema.Union([
|
||||
@@ -265,7 +307,9 @@ export interface ParserState {
|
||||
readonly finishReason?: FinishReasonDetails
|
||||
readonly lifecycle: Lifecycle.State
|
||||
readonly reasoningField?: string
|
||||
readonly reasoningDetails: Array<unknown>
|
||||
/** A scalar reasoning field (`reasoning_content`, ...) has carried text in this stream. */
|
||||
readonly reasoningTextObserved: boolean
|
||||
readonly reasoningDetails: Array<ReasoningDetail>
|
||||
readonly reasoningDetailsObserved: boolean
|
||||
readonly reasoningEmitted: boolean
|
||||
readonly latestToolIndex?: number
|
||||
@@ -320,13 +364,19 @@ const lowerToolCall = (part: ToolCallPart, options: LoweringOptions): OpenAIChat
|
||||
})
|
||||
|
||||
const lowerMedia = Effect.fn("OpenAIChat.lowerMedia")(function* (part: MediaPart) {
|
||||
const media = ProviderShared.normalizeMedia(part)
|
||||
if (!media.mime.startsWith("image/"))
|
||||
return yield* ProviderShared.invalidRequest(`OpenAI Chat does not support media type ${part.mediaType}`)
|
||||
// Chat Completions accepts PDFs, and no other documents, as inline `file` parts; file URLs are not supported.
|
||||
if (part.media.mediaType.toLowerCase() === "application/pdf")
|
||||
return {
|
||||
type: "file" as const,
|
||||
file: {
|
||||
filename: part.filename ?? "document.pdf",
|
||||
file_data: (yield* ProviderShared.requireInlineMedia("OpenAI Chat", part.media)).dataUrl,
|
||||
},
|
||||
}
|
||||
if (part.media.kind !== "image")
|
||||
return yield* ProviderShared.invalidRequest(`OpenAI Chat does not support media type ${part.media.mediaType}`)
|
||||
const url =
|
||||
typeof part.data === "string" && (part.data.startsWith("https://") || part.data.startsWith("http://"))
|
||||
? part.data
|
||||
: media.dataUrl
|
||||
ProviderShared.mediaUrl(part.media) ?? (yield* ProviderShared.requireInlineMedia("OpenAI Chat", part.media)).dataUrl
|
||||
return { type: "image_url" as const, image_url: { url } }
|
||||
})
|
||||
|
||||
@@ -344,10 +394,21 @@ const reasoningDetails = (parts: ReadonlyArray<ReasoningPart>, native: unknown,
|
||||
return Array.isArray(details) ? details : []
|
||||
})
|
||||
if (parts.some((part) => Array.isArray(part.providerMetadata?.[providerMetadataKey]?.reasoningDetails)))
|
||||
return observed
|
||||
if (isRecord(native) && Array.isArray(native.reasoning_details)) return native.reasoning_details
|
||||
return knownReasoningDetails(observed).map(lowerReasoningDetail)
|
||||
if (isRecord(native) && Array.isArray(native.reasoning_details))
|
||||
return knownReasoningDetails(native.reasoning_details).map(lowerReasoningDetail)
|
||||
}
|
||||
|
||||
// Kimi rejects its stream-only `index` on requests
|
||||
// ("the reasoning_details ... must not contain streaming index").
|
||||
const lowerReasoningDetail = (detail: ReasoningDetail) => {
|
||||
if (detail.type === "summary") return { type: detail.type, summary: detail.summary }
|
||||
if (detail.type === "encrypted") return { type: detail.type, encrypted: detail.encrypted }
|
||||
return detail
|
||||
}
|
||||
|
||||
const isKimiDetail = (detail: { readonly type: string }) => detail.type === "summary" || detail.type === "encrypted"
|
||||
|
||||
const lowerUserMessage = Effect.fn("OpenAIChat.lowerUserMessage")(function* (
|
||||
message: OpenAIChatRequestMessage,
|
||||
options: LoweringOptions,
|
||||
@@ -413,6 +474,9 @@ const lowerAssistantMessage = Effect.fn("OpenAIChat.lowerAssistantMessage")(func
|
||||
if (observedField !== undefined) return observedField
|
||||
if (nativeReasoning !== undefined) return "reasoning_content"
|
||||
if (!fullyStructured || requireReasoning) return "reasoning_content"
|
||||
// Kimi always expects `reasoning_content` on replayed assistant messages,
|
||||
// even when thinking arrived only through structured details.
|
||||
if (details?.some(isKimiDetail)) return "reasoning_content"
|
||||
})()
|
||||
const reasoningText = (() => {
|
||||
if (configuredField !== undefined)
|
||||
@@ -438,7 +502,7 @@ const lowerToolMessages = Effect.fn("OpenAIChat.lowerToolMessages")(function* (
|
||||
options: LoweringOptions,
|
||||
) {
|
||||
const messages: OpenAIChatMessage[] = []
|
||||
const images: Array<Schema.Schema.Type<typeof OpenAIChatUserContent>> = []
|
||||
const attachments: Array<Schema.Schema.Type<typeof OpenAIChatUserContent>> = []
|
||||
for (const part of message.content) {
|
||||
if (!ProviderShared.supportsContent(part, ["tool-result"]))
|
||||
return yield* ProviderShared.unsupportedContent("OpenAI Chat", "tool", ["tool-result"])
|
||||
@@ -460,13 +524,9 @@ const lowerToolMessages = Effect.fn("OpenAIChat.lowerToolMessages")(function* (
|
||||
cache_control: options.cacheControl?.(part.cache),
|
||||
})
|
||||
const files = content.filter((item) => item.type === "file")
|
||||
images.push(
|
||||
...(yield* Effect.forEach(files, (item) =>
|
||||
lowerMedia({ type: "media", mediaType: item.mime, data: item.uri, filename: item.name }),
|
||||
)),
|
||||
)
|
||||
attachments.push(...(yield* Effect.forEach(files, (item) => lowerMedia(ProviderShared.toolFileMedia(item)))))
|
||||
}
|
||||
return { messages, images }
|
||||
return { messages, attachments }
|
||||
})
|
||||
|
||||
const lowerMessage = Effect.fn("OpenAIChat.lowerMessage")(function* (
|
||||
@@ -527,21 +587,21 @@ const lowerMessages = Effect.fn("OpenAIChat.lowerMessages")(function* (request:
|
||||
if (requireAssistantAfterTool && messages.at(-1)?.role === "tool")
|
||||
messages.push({ role: "assistant", content: "Done." })
|
||||
}
|
||||
const pendingImages: Array<Schema.Schema.Type<typeof OpenAIChatUserContent>> = []
|
||||
const flushImages = () => {
|
||||
if (pendingImages.length === 0) return
|
||||
const pendingAttachments: Array<Schema.Schema.Type<typeof OpenAIChatUserContent>> = []
|
||||
const flushAttachments = () => {
|
||||
if (pendingAttachments.length === 0) return
|
||||
bridgeTools()
|
||||
messages.push({ role: "user", content: pendingImages.splice(0) })
|
||||
messages.push({ role: "user", content: pendingAttachments.splice(0) })
|
||||
}
|
||||
for (const message of request.messages) {
|
||||
if (message.role === "user") bridgeTools()
|
||||
if (message.role === "system") {
|
||||
const part = yield* ProviderShared.wrappedSystemUpdate("OpenAI Chat", message)
|
||||
if (pendingImages.length > 0) {
|
||||
if (pendingAttachments.length > 0) {
|
||||
messages.push({
|
||||
role: "user",
|
||||
content: [
|
||||
...pendingImages.splice(0),
|
||||
...pendingAttachments.splice(0),
|
||||
{ type: "text", text: part.text, cache_control: options.cacheControl?.(part.cache) },
|
||||
],
|
||||
})
|
||||
@@ -585,13 +645,13 @@ const lowerMessages = Effect.fn("OpenAIChat.lowerMessages")(function* (request:
|
||||
if (message.role === "tool") {
|
||||
const lowered = yield* lowerToolMessages(message, lowering)
|
||||
messages.push(...lowered.messages)
|
||||
pendingImages.push(...lowered.images)
|
||||
pendingAttachments.push(...lowered.attachments)
|
||||
continue
|
||||
}
|
||||
flushImages()
|
||||
flushAttachments()
|
||||
messages.push(...(yield* lowerMessage(message, reasoningField, requireReasoning, lowering)))
|
||||
}
|
||||
flushImages()
|
||||
flushAttachments()
|
||||
return messages
|
||||
})
|
||||
|
||||
@@ -718,7 +778,8 @@ const lowerOptions = (request: LLMRequest, supportsStore: boolean) => {
|
||||
// Default off: strict providers 400 on unknown body fields, so only send
|
||||
// the key where compatibility explicitly allows it. Header-based affinity
|
||||
// (x-session-affinity, x-grok-conv-id, ...) is unaffected.
|
||||
const cacheKey = (request.model.compatibility?.supportsPromptCacheKey ?? false) ? ProviderShared.promptCacheKey(request) : undefined
|
||||
const cacheKey =
|
||||
(request.model.compatibility?.supportsPromptCacheKey ?? false) ? ProviderShared.promptCacheKey(request) : undefined
|
||||
return {
|
||||
...(supportsStore && options.store !== undefined ? { store: options.store } : {}),
|
||||
// For providers that support `store`, ensure stateless `store:false` is sent
|
||||
@@ -887,44 +948,74 @@ const reasoningDelta = (
|
||||
return undefined
|
||||
}
|
||||
|
||||
const detailText = (details: ReadonlyArray<unknown>) => {
|
||||
const detailText = (details: ReadonlyArray<ReasoningDetail>, hideKimiSummary: boolean) => {
|
||||
const text = details.flatMap((detail) => {
|
||||
if (!isRecord(detail)) return []
|
||||
if (detail.type === "reasoning.text" && typeof detail.text === "string" && detail.text) return [detail.text]
|
||||
if (detail.type === "reasoning.summary" && typeof detail.summary === "string" && detail.summary)
|
||||
return [detail.summary]
|
||||
if (detail.type === "reasoning.text") return detail.text ? [detail.text] : []
|
||||
if (detail.type === "reasoning.summary") return detail.summary ? [detail.summary] : []
|
||||
// Kimi streams the full thinking through `reasoning_content` and a separate
|
||||
// summary through details; show the summary only when nothing else does.
|
||||
if (detail.type === "summary") return detail.summary && !hideKimiSummary ? [detail.summary] : []
|
||||
return []
|
||||
})
|
||||
if (text.length > 0) return text.join("")
|
||||
}
|
||||
|
||||
const appendReasoningDetails = (result: Array<unknown>, details: ReadonlyArray<unknown>) => {
|
||||
const appendReasoningDetails = (result: Array<ReasoningDetail>, details: ReadonlyArray<ReasoningDetail>) => {
|
||||
for (const detail of details) {
|
||||
const previous = result.at(-1)
|
||||
if (
|
||||
!isRecord(previous) ||
|
||||
previous.type !== "reasoning.text" ||
|
||||
!isRecord(detail) ||
|
||||
detail.type !== "reasoning.text" ||
|
||||
conflictingReasoningTextDetails(previous, detail)
|
||||
) {
|
||||
const merged = previous === undefined ? undefined : mergeReasoningDetails(previous, detail)
|
||||
if (merged === undefined) {
|
||||
result.push(detail)
|
||||
continue
|
||||
}
|
||||
result[result.length - 1] = {
|
||||
...previous,
|
||||
...Object.fromEntries(Object.entries(detail).filter((entry) => entry[1] !== undefined)),
|
||||
text: `${typeof previous.text === "string" ? previous.text : ""}${typeof detail.text === "string" ? detail.text : ""}`,
|
||||
signature: mergeDetailValue(previous.signature, detail.signature),
|
||||
format: mergeDetailValue(previous.format, detail.format),
|
||||
}
|
||||
result[result.length - 1] = merged
|
||||
}
|
||||
}
|
||||
|
||||
const mergeDetailValue = (previous: unknown, current: unknown) =>
|
||||
// Consecutive text or summary deltas of the same kind accumulate into one
|
||||
// entry; encrypted entries are opaque and never merge.
|
||||
const mergeReasoningDetails = (previous: ReasoningDetail, detail: ReasoningDetail): ReasoningDetail | undefined => {
|
||||
if (conflictingReasoningDetails(previous, detail)) return undefined
|
||||
if (previous.type === "reasoning.text" && detail.type === "reasoning.text")
|
||||
return {
|
||||
...previous,
|
||||
...detail,
|
||||
text: `${previous.text ?? ""}${detail.text ?? ""}`,
|
||||
...mergeDetailIdentity(previous, detail),
|
||||
}
|
||||
if (previous.type === "reasoning.summary" && detail.type === "reasoning.summary")
|
||||
return {
|
||||
...previous,
|
||||
...detail,
|
||||
summary: `${previous.summary ?? ""}${detail.summary ?? ""}`,
|
||||
...mergeDetailIdentity(previous, detail),
|
||||
}
|
||||
if (previous.type === "summary" && detail.type === "summary")
|
||||
return { ...previous, ...detail, summary: previous.summary + detail.summary }
|
||||
}
|
||||
|
||||
type DetailIdentity = {
|
||||
readonly id?: string | null
|
||||
readonly index?: number
|
||||
readonly format?: string
|
||||
readonly signature?: string | null
|
||||
}
|
||||
|
||||
// The first non-empty signature and format win; a later delta may carry the
|
||||
// signature for text that streamed earlier.
|
||||
const mergeDetailIdentity = (previous: DetailIdentity, current: DetailIdentity) => {
|
||||
const signature = mergeDetailValue(previous.signature, current.signature)
|
||||
const format = mergeDetailValue(previous.format, current.format)
|
||||
return {
|
||||
...(signature === undefined ? {} : { signature }),
|
||||
...(format === undefined ? {} : { format }),
|
||||
}
|
||||
}
|
||||
|
||||
const mergeDetailValue = <T>(previous: T | undefined, current: T | undefined) =>
|
||||
previous || current || (previous !== undefined ? previous : current)
|
||||
|
||||
const conflictingReasoningTextDetails = (previous: Record<string, unknown>, current: Record<string, unknown>) =>
|
||||
const conflictingReasoningDetails = (previous: DetailIdentity, current: DetailIdentity) =>
|
||||
conflictingDetailValue(previous.id, current.id) ||
|
||||
conflictingDetailValue(previous.index, current.index) ||
|
||||
conflictingDetailValue(previous.format, current.format) ||
|
||||
@@ -936,7 +1027,7 @@ const conflictingDetailValue = (previous: unknown, current: unknown) =>
|
||||
const reasoningMetadata = (
|
||||
providerMetadataKey: string,
|
||||
field: ParserState["reasoningField"],
|
||||
details?: ReadonlyArray<unknown>,
|
||||
details?: ReadonlyArray<ReasoningDetail>,
|
||||
) => ({
|
||||
[providerMetadataKey]: {
|
||||
...(field ? { reasoningField: field } : {}),
|
||||
@@ -999,11 +1090,16 @@ const step = (state: ParserState, event: OpenAIChatEvent) =>
|
||||
}
|
||||
|
||||
const reasoningField = state.reasoningField ?? reasoning?.field
|
||||
const detailDelta = Array.isArray(delta?.reasoning_details) ? delta.reasoning_details : undefined
|
||||
const reasoningTextObserved = state.reasoningTextObserved || reasoning !== undefined
|
||||
const detailDelta = Array.isArray(delta?.reasoning_details)
|
||||
? knownReasoningDetails(delta.reasoning_details)
|
||||
: undefined
|
||||
if (detailDelta !== undefined) appendReasoningDetails(state.reasoningDetails, detailDelta)
|
||||
const reasoningDetailsObserved = state.reasoningDetailsObserved || detailDelta !== undefined
|
||||
const deltaMetadata = reasoningMetadata(state.providerMetadataKey, reasoningField)
|
||||
const text = detailDelta?.length ? (detailText(detailDelta) ?? reasoning?.text) : reasoning?.text
|
||||
const text = detailDelta?.length
|
||||
? (detailText(detailDelta, reasoningTextObserved) ?? reasoning?.text)
|
||||
: reasoning?.text
|
||||
if (text !== undefined) lifecycle = Lifecycle.reasoningDelta(lifecycle, events, "reasoning-0", text, deltaMetadata)
|
||||
else if (
|
||||
reasoningDetailsObserved &&
|
||||
@@ -1099,6 +1195,7 @@ const step = (state: ParserState, event: OpenAIChatEvent) =>
|
||||
finishReason,
|
||||
lifecycle,
|
||||
reasoningField,
|
||||
reasoningTextObserved,
|
||||
reasoningDetails: state.reasoningDetails,
|
||||
reasoningDetailsObserved,
|
||||
reasoningEmitted,
|
||||
@@ -1179,6 +1276,7 @@ export const protocol = Protocol.make({
|
||||
toolCallEvents: [],
|
||||
lifecycle: Lifecycle.initial(),
|
||||
reasoningField: request.model.compatibility?.reasoningField,
|
||||
reasoningTextObserved: false,
|
||||
reasoningDetails: [],
|
||||
reasoningDetailsObserved: false,
|
||||
reasoningEmitted: false,
|
||||
|
||||
@@ -1,43 +1,39 @@
|
||||
import { Effect, Encoding, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"
|
||||
import {
|
||||
ImageModel,
|
||||
GeneratedImage,
|
||||
ImageResponse,
|
||||
type ImageInput,
|
||||
type ImageRequestFor,
|
||||
type ImageRoute,
|
||||
} from "../image.js"
|
||||
import { Auth, type Definition as AuthDefinition } from "../route/auth.js"
|
||||
import { Usage, mergeHttpOptions, mergeJsonRecords, type HttpOptions } from "../schema/index.js"
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { ProviderShared } from "./shared.js"
|
||||
import { ImageInputs } from "./utils/image-input.js"
|
||||
import { OpenAIImage } from "./utils/openai-image.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "openai-images"
|
||||
const NAME = "OpenAI Images"
|
||||
const PROVIDER = ProviderID.make("openai")
|
||||
export const DEFAULT_BASE_URL = "https://api.openai.com/v1"
|
||||
export const PATH = "/images/generations"
|
||||
export const EDIT_PATH = "/images/edits"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type OpenAIImageString<Known extends string> = Known | (string & {})
|
||||
|
||||
/** Provider-native options. Common fields (`n`, `size`, `format`, `images`, `mask`) live on the request. */
|
||||
export type OpenAIImageOptions = {
|
||||
readonly mask?: ImageInput
|
||||
readonly n?: number
|
||||
readonly size?: OpenAIImageString<
|
||||
"auto" | "256x256" | "512x512" | "1024x1024" | "1536x1024" | "1024x1536" | "1792x1024" | "1024x1792"
|
||||
>
|
||||
readonly quality?: OpenAIImageString<"auto" | "low" | "medium" | "high" | "standard" | "hd">
|
||||
readonly background?: OpenAIImageString<"auto" | "opaque" | "transparent">
|
||||
readonly moderation?: OpenAIImageString<"auto" | "low">
|
||||
readonly outputFormat?: OpenAIImageString<"png" | "jpeg" | "webp">
|
||||
readonly outputCompression?: number
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type OpenAIImageBody = Record<string, unknown> & {
|
||||
readonly model: string
|
||||
readonly prompt: string
|
||||
}
|
||||
export type Request = ImageRequestFor<OpenAIImageOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Response schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const OpenAIImageResponse = Schema.Struct({
|
||||
data: Schema.Array(
|
||||
@@ -59,196 +55,143 @@ const OpenAIImageResponse = Schema.Struct({
|
||||
),
|
||||
})
|
||||
|
||||
export interface ModelInput {
|
||||
readonly id: string
|
||||
readonly auth: AuthDefinition
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Multipart field names the route owns; `http.body` overlays cannot smuggle replacements for them. */
|
||||
const RESERVED_FORM_FIELDS = new Set(["model", "prompt", "image", "image[]", "images", "mask"])
|
||||
|
||||
const nativeOptions = (options: OpenAIImageOptions | undefined) => {
|
||||
if (!options) return undefined
|
||||
const { mask: _, outputFormat, outputCompression, ...native } = options
|
||||
return {
|
||||
output_format: outputFormat,
|
||||
output_compression: outputCompression,
|
||||
...native,
|
||||
}
|
||||
const { outputCompression, ...native } = options
|
||||
return { output_compression: outputCompression, ...native }
|
||||
}
|
||||
|
||||
const applyQuery = (url: string, query: Record<string, string> | undefined) => {
|
||||
if (!query) return url
|
||||
const next = new URL(url)
|
||||
Object.entries(query).forEach(([key, value]) => next.searchParams.set(key, value))
|
||||
return next.toString()
|
||||
}
|
||||
const isEdit = (request: Request) => (request.images?.length ?? 0) > 0
|
||||
|
||||
export const model = (input: ModelInput) => {
|
||||
const route: ImageRoute<OpenAIImageOptions> = {
|
||||
id: ADAPTER,
|
||||
generate: Effect.fn("OpenAIImages.generate")(function* (request: ImageRequestFor<OpenAIImageOptions>, execute) {
|
||||
const mask = request.options?.mask
|
||||
if (mask !== undefined && (request.images?.length ?? 0) === 0)
|
||||
return yield* ImageInputs.invalid("An OpenAI image mask requires at least one input image")
|
||||
const http = mergeHttpOptions(request.model.http, request.http)
|
||||
const sourceImages = request.images ?? []
|
||||
const multipartImages = yield* Effect.forEach(sourceImages, (image) => {
|
||||
if (image.type === "bytes") return Effect.succeed({ data: image.data, mediaType: image.mediaType })
|
||||
if (image.type === "url") return ImageInputs.decodeDataUrl(image.url)
|
||||
return Effect.undefined
|
||||
})
|
||||
const multipartMask =
|
||||
mask === undefined
|
||||
? undefined
|
||||
: mask.type === "bytes"
|
||||
? { data: mask.data, mediaType: mask.mediaType }
|
||||
: mask.type === "url"
|
||||
? yield* ImageInputs.decodeDataUrl(mask.url)
|
||||
: undefined
|
||||
const useMultipart =
|
||||
sourceImages.length > 0 &&
|
||||
multipartImages.every((image) => image !== undefined) &&
|
||||
(mask === undefined || multipartMask !== undefined)
|
||||
const path = sourceImages.length === 0 ? PATH : EDIT_PATH
|
||||
const url = applyQuery(`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${path}`, http?.query)
|
||||
const isInline = (asset: Media.Asset) => asset.inline() !== undefined
|
||||
|
||||
if (useMultipart) {
|
||||
const form = new FormData()
|
||||
form.append("model", request.model.id)
|
||||
form.append("prompt", request.prompt)
|
||||
Object.entries(mergeJsonRecords(nativeOptions(request.options), http?.body) ?? {}).forEach(([key, value]) => {
|
||||
if (["model", "prompt", "image", "image[]", "images", "mask"].includes(key)) return
|
||||
form.append(key, typeof value === "string" ? value : ProviderShared.encodeJson(value))
|
||||
})
|
||||
multipartImages.forEach((image, index) => {
|
||||
if (image === undefined) return
|
||||
form.append("image[]", imageBlob(image.data, image.mediaType), `image-${index}`)
|
||||
})
|
||||
if (multipartMask !== undefined)
|
||||
form.append("mask", imageBlob(multipartMask.data, multipartMask.mediaType), "mask")
|
||||
const headers = yield* Auth.toEffect(input.auth)({
|
||||
request,
|
||||
method: "POST",
|
||||
url,
|
||||
body: "[multipart/form-data]",
|
||||
headers: Headers.remove(Headers.fromInput({ ...input.headers, ...http?.headers }), "content-type"),
|
||||
})
|
||||
const response = yield* execute(
|
||||
HttpClientRequest.post(url).pipe(HttpClientRequest.setHeaders(headers), HttpClientRequest.bodyFormData(form)),
|
||||
)
|
||||
return yield* parseResponse(response, request.options, http?.body)
|
||||
}
|
||||
|
||||
const references = sourceImages.map((image) => {
|
||||
if (image.type === "bytes") return { image_url: ImageInputs.dataUrl(image) }
|
||||
if (image.type === "url") return { image_url: image.url }
|
||||
if (image.type === "file-id") return { file_id: image.id }
|
||||
return undefined
|
||||
})
|
||||
if (references.some((image) => image === undefined))
|
||||
return yield* ImageInputs.invalid("OpenAI Images accepts image URLs, data URLs, bytes, and file IDs")
|
||||
const maskReference =
|
||||
mask === undefined
|
||||
? undefined
|
||||
: mask.type === "bytes"
|
||||
? { image_url: ImageInputs.dataUrl(mask) }
|
||||
: mask.type === "url"
|
||||
? { image_url: mask.url }
|
||||
: mask.type === "file-id"
|
||||
? { file_id: mask.id }
|
||||
: undefined
|
||||
if (mask !== undefined && maskReference === undefined)
|
||||
return yield* ImageInputs.invalid("OpenAI Images accepts masks as URLs, data URLs, bytes, or file IDs")
|
||||
const requestBody = mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
images: references.length === 0 ? undefined : references,
|
||||
mask: maskReference,
|
||||
},
|
||||
nativeOptions(request.options),
|
||||
http?.body,
|
||||
) as OpenAIImageBody
|
||||
const text = ProviderShared.encodeJson(requestBody)
|
||||
const headers = yield* Auth.toEffect(input.auth)({
|
||||
request,
|
||||
method: "POST",
|
||||
url,
|
||||
body: text,
|
||||
headers: Headers.fromInput({ ...input.headers, ...http?.headers }),
|
||||
})
|
||||
const response = yield* execute(
|
||||
HttpClientRequest.post(url).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(text, "application/json"),
|
||||
),
|
||||
)
|
||||
return yield* parseResponse(response, request.options, http?.body)
|
||||
}),
|
||||
}
|
||||
return ImageModel.make<OpenAIImageOptions>({ id: input.id, provider: "openai", route, http: input.http })
|
||||
}
|
||||
|
||||
const parseResponse = Effect.fn("OpenAIImages.parseResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
options: OpenAIImageOptions | undefined,
|
||||
overlay: Record<string, unknown> | undefined,
|
||||
) {
|
||||
const output = yield* ProviderShared.imageResponse(ADAPTER, "OpenAI Images", response)
|
||||
const decoded = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(OpenAIImageResponse))(output.body).pipe(
|
||||
Effect.mapError((cause) => output.invalid("OpenAI Images returned an invalid response", cause)),
|
||||
const reference = (asset: Media.Asset) =>
|
||||
ProviderShared.mediaReference(asset, PROVIDER, NAME).pipe(
|
||||
Effect.map((item) => (item.type === "ref" ? { file_id: item.value } : { image_url: item.value })),
|
||||
)
|
||||
const requestBody = mergeJsonRecords(nativeOptions(options), overlay)
|
||||
const format =
|
||||
decoded.output_format ?? (typeof requestBody?.output_format === "string" ? requestBody.output_format : "png")
|
||||
|
||||
const fromRequest = Effect.fn("OpenAIImages.fromRequest")(function* (request: Request) {
|
||||
const images = request.images ?? []
|
||||
const mask = request.mask
|
||||
if (mask !== undefined && images.length === 0)
|
||||
return yield* ProviderShared.invalidRequest("An OpenAI image mask requires at least one input image")
|
||||
const fields = mergeJsonRecords(
|
||||
{ n: request.n, size: request.size, output_format: request.format },
|
||||
nativeOptions(request.providerOptions),
|
||||
request.http?.body,
|
||||
)
|
||||
|
||||
// Owned bytes go through multipart edits; remote URLs and file IDs use the JSON edits body instead.
|
||||
if (images.length > 0 && images.every(isInline) && (mask === undefined || isInline(mask))) {
|
||||
const form = new FormData()
|
||||
form.append("model", request.model.id)
|
||||
form.append("prompt", request.prompt)
|
||||
Object.entries(fields ?? {}).forEach(([key, value]) => {
|
||||
if (RESERVED_FORM_FIELDS.has(key)) return
|
||||
form.append(key, typeof value === "string" ? value : ProviderShared.encodeJson(value))
|
||||
})
|
||||
const uploads = yield* Effect.forEach(images, (image) => MediaInput.inlineBytes(ADAPTER, image))
|
||||
uploads.forEach((data, index) =>
|
||||
form.append("image[]", MediaInput.blob(data, images[index].mediaType), `image-${index}`),
|
||||
)
|
||||
if (mask !== undefined)
|
||||
form.append("mask", MediaInput.blob(yield* MediaInput.inlineBytes(ADAPTER, mask), mask.mediaType), "mask")
|
||||
return MediaProtocol.multipart(form)
|
||||
}
|
||||
|
||||
const references = yield* Effect.forEach(images, reference)
|
||||
const maskReference = mask === undefined ? undefined : yield* reference(mask)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
images: references.length === 0 ? undefined : references,
|
||||
mask: maskReference,
|
||||
},
|
||||
fields,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const requestedFormat = (body: MediaProtocol.Body) => {
|
||||
if (body.type === "binary") return undefined
|
||||
const value = body.type === "json" ? body.value.output_format : body.value.get("output_format")
|
||||
return typeof value === "string" ? value : undefined
|
||||
}
|
||||
|
||||
const decodeResponse = Effect.fn("OpenAIImages.decodeResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.DecodeContext<Request>,
|
||||
) {
|
||||
const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, OpenAIImageResponse)(response)
|
||||
const decoded = output.value
|
||||
const format = decoded.output_format ?? requestedFormat(context.body) ?? "png"
|
||||
const mediaType = `image/${format}`
|
||||
const images = yield* Effect.forEach(decoded.data, (item, index) => {
|
||||
const providerMetadata =
|
||||
item.revised_prompt === undefined ? undefined : { openai: { revisedPrompt: item.revised_prompt } }
|
||||
if (item.b64_json)
|
||||
return Effect.fromResult(Encoding.decodeBase64(item.b64_json)).pipe(
|
||||
Effect.mapError((cause) => output.invalid(`OpenAI Images result ${index} contains invalid base64 data`, cause)),
|
||||
Effect.map(
|
||||
(data) =>
|
||||
new GeneratedImage({
|
||||
mediaType: `image/${format}`,
|
||||
data,
|
||||
providerMetadata:
|
||||
item.revised_prompt === undefined ? undefined : { openai: { revisedPrompt: item.revised_prompt } },
|
||||
}),
|
||||
),
|
||||
)
|
||||
if (item.url)
|
||||
return Effect.succeed(
|
||||
new GeneratedImage({
|
||||
mediaType: `image/${format}`,
|
||||
data: item.url,
|
||||
providerMetadata:
|
||||
item.revised_prompt === undefined ? undefined : { openai: { revisedPrompt: item.revised_prompt } },
|
||||
}),
|
||||
)
|
||||
return Effect.fail(output.invalid(`OpenAI Images result ${index} has neither image data nor a URL`))
|
||||
return MediaInput.decodedAsset(output.invalid, `${NAME} result ${index}`, item.b64_json, mediaType, {
|
||||
info: { format },
|
||||
providerMetadata,
|
||||
})
|
||||
if (item.url) return Effect.succeed(Media.url(item.url, { mediaType, info: { format }, providerMetadata }))
|
||||
return Effect.fail(output.invalid(`${NAME} result ${index} has neither image data nor a URL`))
|
||||
})
|
||||
if (images.length === 0) return yield* output.invalid("OpenAI Images returned no images")
|
||||
if (images.length === 0) return yield* output.invalid(`${NAME} returned no images`)
|
||||
return new ImageResponse({
|
||||
images,
|
||||
usage:
|
||||
decoded.usage === undefined
|
||||
? undefined
|
||||
: new Usage({
|
||||
inputTokens: decoded.usage.input_tokens,
|
||||
outputTokens: decoded.usage.output_tokens,
|
||||
totalTokens: decoded.usage.total_tokens,
|
||||
providerMetadata: { openai: decoded.usage },
|
||||
}),
|
||||
: {
|
||||
type: "tokens",
|
||||
input: decoded.usage.input_tokens,
|
||||
output: decoded.usage.output_tokens,
|
||||
total: decoded.usage.total_tokens,
|
||||
details: { openai: decoded.usage },
|
||||
},
|
||||
providerMetadata: { openai: { outputFormat: format } },
|
||||
})
|
||||
})
|
||||
|
||||
const imageBlob = (data: Uint8Array, mediaType: string) => {
|
||||
const buffer = new ArrayBuffer(data.byteLength)
|
||||
new Uint8Array(buffer).set(data)
|
||||
return new Blob([buffer], { type: mediaType })
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.inline<Request, ImageResponse>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["aspectRatio", "seed"],
|
||||
body: { from: fromRequest },
|
||||
response: { decode: decodeResponse },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
ImageModel.fromRoute<OpenAIImageOptions>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => (isEdit(request) ? EDIT_PATH : PATH),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const OpenAIImages = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { Framing } from "../route/framing.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords, type MediaUsage } from "../schema/index.js"
|
||||
import { SpeechModel, type SpeechEvent, type SpeechRequestFor } from "../speech.js"
|
||||
import { SpeechStream } from "./utils/speech-stream.js"
|
||||
|
||||
const ADAPTER = "openai-speech"
|
||||
const NAME = "OpenAI Speech"
|
||||
const PROVIDER = ProviderID.make("openai")
|
||||
export const DEFAULT_BASE_URL = "https://api.openai.com/v1"
|
||||
export const PATH = "/audio/speech"
|
||||
/** `pcm` is raw 24 kHz, 16-bit signed little-endian mono samples without a header. */
|
||||
const PCM_SAMPLE_RATE = 24000
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type OpenAISpeechOptions = Record<string, unknown>
|
||||
|
||||
export type Request = SpeechRequestFor<OpenAISpeechOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Streaming event schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const SpeechStreamEvent = Schema.Union([
|
||||
Schema.Struct({ type: Schema.Literal("speech.audio.delta"), audio: Schema.Uint8ArrayFromBase64 }),
|
||||
Schema.Struct({
|
||||
type: Schema.Literal("speech.audio.done"),
|
||||
usage: Schema.optional(
|
||||
Schema.Struct({
|
||||
input_tokens: Schema.optional(Schema.Number),
|
||||
output_tokens: Schema.optional(Schema.Number),
|
||||
total_tokens: Schema.optional(Schema.Number),
|
||||
}),
|
||||
),
|
||||
}),
|
||||
])
|
||||
|
||||
const decodeEvent = MediaProtocol.decodeFrame(ADAPTER, NAME, SpeechStreamEvent)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface State extends SpeechStream.Audio {
|
||||
readonly done: boolean
|
||||
readonly usage?: MediaUsage
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// `sse` is not supported for `tts-1` or `tts-1-hd`; those models stream the raw audio body instead.
|
||||
const supportsSse = (model: string) => !/^tts-1(-hd)?(-|$)/.test(model)
|
||||
|
||||
const fromRequest = Effect.fn("OpenAISpeech.fromRequest")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
input: request.text,
|
||||
voice: request.voice,
|
||||
instructions: request.instructions,
|
||||
response_format: request.format,
|
||||
speed: request.speed,
|
||||
stream_format: request.mode === "stream" && supportsSse(request.model.id) ? "sse" : undefined,
|
||||
},
|
||||
request.providerOptions,
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const isSse = (body: MediaProtocol.Body) => body.type === "json" && body.value.stream_format === "sse"
|
||||
|
||||
const onEvent = Effect.fn("OpenAISpeech.onEvent")(function* (state: State, frame: string) {
|
||||
const event = yield* decodeEvent(frame)
|
||||
if (event.type === "speech.audio.delta") return SpeechStream.delta(state, event.audio)
|
||||
const usage = event.usage
|
||||
return [
|
||||
{
|
||||
...state,
|
||||
done: true,
|
||||
usage:
|
||||
usage === undefined
|
||||
? undefined
|
||||
: {
|
||||
type: "tokens" as const,
|
||||
input: usage.input_tokens,
|
||||
output: usage.output_tokens,
|
||||
total: usage.total_tokens,
|
||||
details: { openai: usage },
|
||||
},
|
||||
},
|
||||
[],
|
||||
] as const
|
||||
})
|
||||
|
||||
const finish = (state: State, context: MediaProtocol.ResponseContext<Request>) => {
|
||||
if (isSse(context.body) && !state.done) return Effect.fail(MediaProtocol.incomplete(ADAPTER))
|
||||
const format = context.request.format ?? "mp3"
|
||||
return SpeechStream.finish(ADAPTER, state, {
|
||||
...(format === "pcm" ? SpeechStream.pcm("pcm_s16le", PCM_SAMPLE_RATE) : SpeechStream.container(format)),
|
||||
usage: state.usage,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, SpeechEvent, string | Uint8Array, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["language", "timestamps"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes, context) => (isSse(context.body) ? Framing.sse.frame(bytes) : bytes),
|
||||
initial: () => ({ chunks: [], done: false }),
|
||||
step: SpeechStream.step(onEvent),
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
SpeechModel.fromRoute<OpenAISpeechOptions, string | Uint8Array, State>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH },
|
||||
input,
|
||||
)
|
||||
|
||||
export const OpenAISpeech = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,277 @@
|
||||
import { Effect, Schema, Stream } from "effect"
|
||||
import { Framing } from "../route/framing.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords, type MediaUsage } from "../schema/index.js"
|
||||
import {
|
||||
TranscriptionFinishEvent,
|
||||
TranscriptionModel,
|
||||
TranscriptionSegmentEvent,
|
||||
TranscriptionTextDeltaEvent,
|
||||
type TranscriptionEvent,
|
||||
type TranscriptionRequestFor,
|
||||
type TranscriptionSegment,
|
||||
} from "../transcription.js"
|
||||
import { mediaTypeExtension } from "../utils/media-type.js"
|
||||
import { ProviderShared } from "./shared.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "openai-transcription"
|
||||
const NAME = "OpenAI Transcription"
|
||||
const PROVIDER = ProviderID.make("openai")
|
||||
export const DEFAULT_BASE_URL = "https://api.openai.com/v1"
|
||||
export const PATH = "/audio/transcriptions"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type OpenAITranscriptionOptions = {
|
||||
readonly chunking_strategy?:
|
||||
| "auto"
|
||||
| {
|
||||
readonly type: "server_vad"
|
||||
readonly prefix_padding_ms?: number
|
||||
readonly silence_duration_ms?: number
|
||||
readonly threshold?: number
|
||||
}
|
||||
readonly include?: ReadonlyArray<"logprobs">
|
||||
readonly keywords?: ReadonlyArray<string>
|
||||
readonly known_speaker_names?: ReadonlyArray<string>
|
||||
readonly known_speaker_references?: ReadonlyArray<string>
|
||||
readonly temperature?: number
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = TranscriptionRequestFor<OpenAITranscriptionOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 3. Streaming event schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const Segment = Schema.Struct({
|
||||
text: Schema.String,
|
||||
start: Schema.Number,
|
||||
end: Schema.Number,
|
||||
speaker: Schema.optional(Schema.String),
|
||||
})
|
||||
|
||||
const Usage = Schema.Union([
|
||||
Schema.Struct({
|
||||
type: Schema.Literal("tokens"),
|
||||
input_tokens: Schema.optional(Schema.Number),
|
||||
output_tokens: Schema.optional(Schema.Number),
|
||||
total_tokens: Schema.optional(Schema.Number),
|
||||
}),
|
||||
Schema.Struct({ type: Schema.Literal("duration"), seconds: Schema.Number }),
|
||||
])
|
||||
|
||||
const transcriptFields = {
|
||||
text: Schema.String,
|
||||
language: Schema.optional(Schema.String),
|
||||
languages: Schema.optional(Schema.Array(Schema.Struct({ code: Schema.String }))),
|
||||
duration: Schema.optional(Schema.Number),
|
||||
segments: Schema.optional(Schema.Array(Segment)),
|
||||
words: Schema.optional(
|
||||
Schema.Array(Schema.Struct({ word: Schema.String, start: Schema.Number, end: Schema.Number })),
|
||||
),
|
||||
usage: Schema.optional(Usage),
|
||||
}
|
||||
|
||||
const Event = Schema.Union([
|
||||
Schema.Struct({ type: Schema.Literal("transcript.text.delta"), delta: Schema.String }),
|
||||
Schema.Struct({ type: Schema.Literal("transcript.text.segment"), ...Segment.fields }),
|
||||
Schema.Struct({ type: Schema.Literal("transcript.text.done"), ...transcriptFields }),
|
||||
])
|
||||
const Transcript = Schema.Struct(transcriptFields)
|
||||
type Transcript = Schema.Schema.Type<typeof Transcript>
|
||||
|
||||
const decodeEvent = MediaProtocol.decodeFrame(ADAPTER, NAME, Event)
|
||||
const decodeTranscript = MediaProtocol.decodeFrame(ADAPTER, NAME, Transcript)
|
||||
|
||||
type Frame = string | { readonly document: string }
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 4. Parser state
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface State {
|
||||
readonly segments: Array<TranscriptionSegment>
|
||||
readonly transcript?: Transcript
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
interface Capabilities {
|
||||
readonly stream: boolean
|
||||
readonly timestamps: ReadonlyArray<"segment" | "word">
|
||||
readonly diarize: boolean
|
||||
readonly languageField: "language" | "languages"
|
||||
}
|
||||
|
||||
const TRANSCRIBE: Capabilities = { stream: true, timestamps: [], diarize: false, languageField: "language" }
|
||||
|
||||
const capabilities = (model: string): Capabilities => {
|
||||
if (model.startsWith("whisper")) return { ...TRANSCRIBE, stream: false, timestamps: ["segment", "word"] }
|
||||
if (model.includes("diarize")) return { ...TRANSCRIBE, timestamps: ["segment"], diarize: true }
|
||||
// `gpt-transcribe` replaces `language` with `languages[]` and rejects both together.
|
||||
if (model.startsWith("gpt-transcribe")) return { ...TRANSCRIBE, languageField: "languages" }
|
||||
return TRANSCRIBE
|
||||
}
|
||||
|
||||
const unsupported = (operation: string, message: string) =>
|
||||
Effect.fail(ProviderShared.unsupportedOperation({ operation, provider: PROVIDER, route: ADAPTER, message }))
|
||||
|
||||
const validate = (request: MediaProtocol.Addressed<Request>, model: Capabilities) => {
|
||||
const id = request.model.id
|
||||
if (request.mode === "stream" && !model.stream)
|
||||
return unsupported("media.stream", `${id} does not stream; use Transcription.generate`)
|
||||
if (request.diarize === true && !model.diarize)
|
||||
return unsupported("media.diarize", `${id} does not diarize; use gpt-4o-transcribe-diarize`)
|
||||
if (request.prompt !== undefined && model.diarize)
|
||||
return unsupported("media.prompt", `${id} does not accept a prompt`)
|
||||
if (
|
||||
request.timestamps === undefined ||
|
||||
request.timestamps === "none" ||
|
||||
model.timestamps.includes(request.timestamps)
|
||||
)
|
||||
return Effect.void
|
||||
return unsupported("media.timestamps", `${id} does not return ${request.timestamps} timestamps`)
|
||||
}
|
||||
|
||||
const RESERVED_FORM_FIELDS = new Set([
|
||||
"file",
|
||||
"model",
|
||||
"prompt",
|
||||
"language",
|
||||
"response_format",
|
||||
"timestamp_granularities",
|
||||
"stream",
|
||||
])
|
||||
|
||||
const appendField = (form: FormData, key: string, value: unknown) => {
|
||||
if (Array.isArray(value)) return value.forEach((item) => form.append(`${key}[]`, String(item)))
|
||||
form.append(key, typeof value === "object" && value !== null ? ProviderShared.encodeJson(value) : String(value))
|
||||
}
|
||||
|
||||
const fromRequest = Effect.fn("OpenAITranscription.fromRequest")(function* (request: MediaProtocol.Addressed<Request>) {
|
||||
const model = capabilities(request.model.id)
|
||||
yield* validate(request, model)
|
||||
// The API detects the audio format from the upload's filename extension.
|
||||
const extension = mediaTypeExtension(request.audio.mediaType)
|
||||
if (extension === undefined)
|
||||
return yield* ProviderShared.invalidRequest(
|
||||
`${NAME} cannot name a ${request.audio.mediaType} upload; send mp3, mp4, m4a, wav, webm, ogg, or flac audio`,
|
||||
)
|
||||
const audio = yield* MediaInput.inlineBytes(ADAPTER, request.audio)
|
||||
const responseFormat = model.diarize
|
||||
? "diarized_json"
|
||||
: request.timestamps === undefined || request.timestamps === "none"
|
||||
? undefined
|
||||
: "verbose_json"
|
||||
const native = Object.entries(mergeJsonRecords(request.providerOptions, request.http?.body) ?? {}).filter(
|
||||
([key]) => !RESERVED_FORM_FIELDS.has(key),
|
||||
)
|
||||
const fields = mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
language: model.languageField === "language" ? request.language : undefined,
|
||||
languages: model.languageField === "languages" && request.language !== undefined ? [request.language] : undefined,
|
||||
prompt: request.prompt,
|
||||
response_format: responseFormat,
|
||||
timestamp_granularities: responseFormat === "verbose_json" ? [request.timestamps] : undefined,
|
||||
// Diarizing audio longer than 30 seconds requires a chunking strategy.
|
||||
chunking_strategy: model.diarize ? "auto" : undefined,
|
||||
stream: request.mode === "stream" ? true : undefined,
|
||||
},
|
||||
Object.fromEntries(native),
|
||||
)
|
||||
const form = new FormData()
|
||||
form.append("file", MediaInput.blob(audio, request.audio.mediaType), `audio.${extension}`)
|
||||
Object.entries(fields ?? {}).forEach(([key, value]) => appendField(form, key, value))
|
||||
return MediaProtocol.multipart(form)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const segment = (value: Schema.Schema.Type<typeof Segment>): TranscriptionSegment => ({
|
||||
text: value.text.trim(),
|
||||
startSeconds: value.start,
|
||||
endSeconds: value.end,
|
||||
speaker: value.speaker,
|
||||
})
|
||||
|
||||
const onEvent = Effect.fn("OpenAITranscription.onEvent")(function* (state: State, frame: string) {
|
||||
const event = yield* decodeEvent(frame)
|
||||
if (event.type === "transcript.text.done") return [{ ...state, transcript: event }, []] as const
|
||||
if (event.type === "transcript.text.delta")
|
||||
return [state, event.delta.length === 0 ? [] : [TranscriptionTextDeltaEvent.make({ delta: event.delta })]] as const
|
||||
const next = segment(event)
|
||||
state.segments.push(next)
|
||||
return [state, [TranscriptionSegmentEvent.make({ segment: next })]] as const
|
||||
})
|
||||
|
||||
const step = (state: State, frame: Frame) =>
|
||||
typeof frame === "string"
|
||||
? onEvent(state, frame)
|
||||
: decodeTranscript(frame.document).pipe(Effect.map((transcript) => [{ ...state, transcript }, []] as const))
|
||||
|
||||
const usage = (value: Transcript["usage"]): MediaUsage | undefined => {
|
||||
if (value === undefined) return undefined
|
||||
if (value.type === "duration") return { type: "seconds", seconds: value.seconds }
|
||||
return {
|
||||
type: "tokens",
|
||||
input: value.input_tokens,
|
||||
output: value.output_tokens,
|
||||
total: value.total_tokens,
|
||||
details: { openai: value },
|
||||
}
|
||||
}
|
||||
|
||||
const finish = (state: State) => {
|
||||
const transcript = state.transcript
|
||||
if (transcript === undefined) return Effect.fail(MediaProtocol.incomplete(ADAPTER))
|
||||
const segments = transcript.segments?.map(segment) ?? state.segments
|
||||
return Effect.succeed([
|
||||
TranscriptionFinishEvent.make({
|
||||
text: transcript.text,
|
||||
segments: segments.length === 0 ? undefined : segments,
|
||||
words: transcript.words?.map((word) => ({ text: word.word, startSeconds: word.start, endSeconds: word.end })),
|
||||
language: (transcript.language ?? transcript.languages?.[0]?.code)?.toLowerCase(),
|
||||
durationSeconds: transcript.duration,
|
||||
usage: usage(transcript.usage),
|
||||
}),
|
||||
])
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.stream<Request, TranscriptionEvent, Frame, State>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["speakers"],
|
||||
body: { from: fromRequest },
|
||||
frames: (bytes, context) =>
|
||||
context.request.mode === "stream"
|
||||
? Framing.sse.frame(bytes)
|
||||
: Framing.document.frame(bytes).pipe(Stream.map((document) => ({ document }))),
|
||||
initial: () => ({ segments: [] }),
|
||||
step,
|
||||
finish,
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
TranscriptionModel.fromRoute<OpenAITranscriptionOptions, Frame, State>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH },
|
||||
input,
|
||||
)
|
||||
|
||||
export const OpenAITranscription = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -0,0 +1,204 @@
|
||||
import { Duration, Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import type { Status } from "../generation.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { VideoModel, VideoResponse, type VideoRequestFor } from "../video.js"
|
||||
import { ProviderShared, optionalArray, optionalNull } from "./shared.js"
|
||||
|
||||
const ADAPTER = "runway-video"
|
||||
const NAME = "Runway"
|
||||
const PROVIDER = ProviderID.make("runway")
|
||||
export const DEFAULT_BASE_URL = "https://api.dev.runwayml.com/v1"
|
||||
/** Every Runway request must pin the API version. */
|
||||
export const API_VERSION = "2024-11-06"
|
||||
export const TEXT_TO_VIDEO_PATH = "/text_to_video"
|
||||
export const IMAGE_TO_VIDEO_PATH = "/image_to_video"
|
||||
export const VIDEO_TO_VIDEO_PATH = "/video_to_video"
|
||||
export const TASKS_PATH = "/tasks"
|
||||
/** Output URLs are valid for 24–48 hours; the asset carries the conservative bound. */
|
||||
const OUTPUT_RETENTION = Duration.hours(24)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type RunwayVideoString<Known extends string> = Known | (string & {})
|
||||
|
||||
/**
|
||||
* Provider-native options. Common fields lower to Runway's names: `aspectRatio` → `ratio` (Runway expects pixel
|
||||
* ratios such as `1280:720` for most models), `durationSeconds` → `duration`, `audio`, `negativePrompt`,
|
||||
* `resolution`, `references`, and `frames` → `promptImage`.
|
||||
*/
|
||||
export type RunwayVideoOptions = {
|
||||
readonly contentModeration?: { readonly publicFigureThreshold?: RunwayVideoString<"auto" | "low"> }
|
||||
readonly outputFormat?: RunwayVideoString<"mp4" | "prores" | "png_sequence">
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = VideoRequestFor<RunwayVideoOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Token and response schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const Token = Schema.Struct({ taskID: Schema.String })
|
||||
export type Token = Schema.Schema.Type<typeof Token>
|
||||
|
||||
const Cost = Schema.Struct({ credits: Schema.Number })
|
||||
|
||||
const StartResponse = Schema.Struct({ id: Schema.String })
|
||||
|
||||
const Task = Schema.Struct({
|
||||
status: Schema.String,
|
||||
progress: optionalNull(Schema.Number),
|
||||
output: optionalArray(Schema.String),
|
||||
failure: optionalNull(Schema.String),
|
||||
failureCode: optionalNull(Schema.String),
|
||||
cost: Schema.optional(Cost),
|
||||
estimatedCost: Schema.optional(Cost),
|
||||
})
|
||||
|
||||
const STATUS = {
|
||||
PENDING: "queued",
|
||||
THROTTLED: "queued",
|
||||
RUNNING: "running",
|
||||
SUCCEEDED: "completed",
|
||||
FAILED: "failed",
|
||||
CANCELLED: "cancelled",
|
||||
} as const satisfies Record<string, Status>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Runway accepts HTTPS URLs, `runway://` upload URIs, and data URIs, all as one string.
|
||||
const mediaUri = (asset: Media.Asset) =>
|
||||
ProviderShared.mediaReference(asset, PROVIDER, NAME).pipe(Effect.map((reference) => reference.value))
|
||||
|
||||
const fromRequest = Effect.fn("RunwayVideo.fromRequest")(function* (request: Request) {
|
||||
const first = request.frames?.first === undefined ? undefined : yield* mediaUri(request.frames.first)
|
||||
const last = request.frames?.last === undefined ? undefined : yield* mediaUri(request.frames.last)
|
||||
const promptImage = [
|
||||
...(first === undefined ? [] : [{ uri: first, position: "first" }]),
|
||||
...(last === undefined ? [] : [{ uri: last, position: "last" }]),
|
||||
]
|
||||
const videoUri = request.video === undefined ? undefined : yield* mediaUri(request.video)
|
||||
const references = yield* Effect.forEach(request.references ?? [], (asset) =>
|
||||
mediaUri(asset).pipe(Effect.map((uri) => ({ uri }))),
|
||||
)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
promptText: request.prompt,
|
||||
promptImage: promptImage.length === 0 ? undefined : promptImage,
|
||||
videoUri,
|
||||
references: references.length === 0 ? undefined : references,
|
||||
ratio: request.aspectRatio,
|
||||
duration: request.durationSeconds,
|
||||
resolution: request.resolution,
|
||||
audio: request.audio,
|
||||
negativePrompt: request.negativePrompt,
|
||||
seed: request.seed,
|
||||
},
|
||||
request.providerOptions,
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeStart = MediaProtocol.decodeStarted(ADAPTER, NAME, StartResponse, (value) => ({
|
||||
token: { taskID: value.id },
|
||||
snapshot: { id: value.id, status: "queued" },
|
||||
}))
|
||||
|
||||
const decodeTask = MediaProtocol.decodeJson(ADAPTER, NAME, Task)
|
||||
|
||||
const decodeStatus = Effect.fn("RunwayVideo.decodeStatus")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeTask(response)
|
||||
const status = yield* MediaProtocol.status(STATUS, output.value.status, output)
|
||||
return { id: context.token.taskID, status, progress: output.value.progress ?? undefined }
|
||||
})
|
||||
|
||||
const decodeResult = Effect.fn("RunwayVideo.decodeResult")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeTask(response)
|
||||
const task = output.value
|
||||
const status = yield* MediaProtocol.status(STATUS, task.status, output)
|
||||
if (status === "failed") {
|
||||
const code = task.failureCode ?? undefined
|
||||
const message = `${NAME} task failed${code === undefined ? "" : ` (${code})`}${task.failure ? `: ${task.failure}` : ""}`
|
||||
// Runway failure codes are dotted paths; every moderation outcome carries a SAFETY segment.
|
||||
if (code !== undefined && /(^|\.)SAFETY(\.|$)/.test(code)) return yield* output.contentPolicy(message)
|
||||
return yield* output.ended("failed", message)
|
||||
}
|
||||
if (status === "cancelled")
|
||||
return yield* output.ended("cancelled", `${NAME} task ${context.token.taskID} was cancelled`)
|
||||
if (status !== "completed") return yield* output.invalid(`${NAME} task ${context.token.taskID} has not finished`)
|
||||
const urls = task.output ?? []
|
||||
if (urls.length === 0) return yield* output.invalid(`${NAME} task succeeded without any output`)
|
||||
return new VideoResponse({
|
||||
videos: yield* Effect.forEach(urls, (url) =>
|
||||
MediaProtocol.expiringUrl(url, OUTPUT_RETENTION, { mediaType: "video/mp4" }),
|
||||
),
|
||||
usage: task.cost === undefined ? undefined : { type: "credits", credits: task.cost.credits },
|
||||
providerMetadata: {
|
||||
runway: {
|
||||
taskId: context.token.taskID,
|
||||
estimatedCredits: task.estimatedCost?.credits,
|
||||
},
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const taskPath = (token: Token) => `${TASKS_PATH}/${token.taskID}`
|
||||
|
||||
export const protocol = MediaProtocol.queued<Request, VideoResponse, Token>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
token: Token,
|
||||
unsupported: ["n"],
|
||||
start: { body: { from: fromRequest }, decode: decodeStart },
|
||||
status: { path: taskPath, decode: decodeStatus },
|
||||
result: { path: taskPath, decode: decodeResult },
|
||||
cancel: { method: "DELETE", path: taskPath },
|
||||
})
|
||||
|
||||
const startPath = (request: Request) => {
|
||||
if (request.video !== undefined) return VIDEO_TO_VIDEO_PATH
|
||||
if (request.frames?.first !== undefined || request.frames?.last !== undefined) return IMAGE_TO_VIDEO_PATH
|
||||
return TEXT_TO_VIDEO_PATH
|
||||
}
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
VideoModel.fromRoute<RunwayVideoOptions, Token>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
headers: { "X-Runway-Version": API_VERSION },
|
||||
path: ({ request }) => startPath(request),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const RunwayVideo = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -1,14 +1,13 @@
|
||||
import { Buffer } from "node:buffer"
|
||||
import { Tool } from "@opencode/schema/tool"
|
||||
import { Effect, Option, Schema, Stream } from "effect"
|
||||
import * as Sse from "effect/unstable/encoding/Sse"
|
||||
import { Headers, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import { Media } from "../media.js"
|
||||
import {
|
||||
InvalidProviderOutputError,
|
||||
InvalidRequestError,
|
||||
UnsupportedOperationError,
|
||||
AIError,
|
||||
HttpContext,
|
||||
LLMRequest,
|
||||
Message,
|
||||
ToolDefinition,
|
||||
@@ -179,24 +178,55 @@ export const wrappedSystemUpdate = Effect.fn("ProviderShared.wrappedSystemUpdate
|
||||
export const parseToolInput = (route: string, name: string, raw: string) =>
|
||||
parseJson(route, raw || "{}", `Invalid JSON input for ${route} tool call ${name}`)
|
||||
|
||||
export interface NormalizedMedia {
|
||||
readonly mime: string
|
||||
readonly base64: string
|
||||
readonly dataUrl: string
|
||||
/** Inline view or a typed `InvalidRequest` for routes that cannot fetch URLs or dereference provider refs. */
|
||||
export const requireInlineMedia = (route: string, asset: Media.Asset): Effect.Effect<Media.Inline, AIError> => {
|
||||
const inline = asset.inline()
|
||||
return inline ? Effect.succeed(inline) : Effect.fail(inlineRequired(route, asset))
|
||||
}
|
||||
|
||||
export const normalizeMedia = (part: MediaPart): NormalizedMedia => {
|
||||
const mime = part.mediaType.toLowerCase()
|
||||
if (typeof part.data !== "string") {
|
||||
const base64 = Buffer.from(part.data).toString("base64")
|
||||
return { mime, base64, dataUrl: `data:${mime};base64,${base64}` }
|
||||
}
|
||||
if (!part.data.startsWith("data:")) return { mime, base64: part.data, dataUrl: `data:${mime};base64,${part.data}` }
|
||||
return { mime, base64: part.data.slice(part.data.indexOf(",") + 1), dataUrl: part.data }
|
||||
export const inlineRequired = (route: string, asset: Media.Asset) =>
|
||||
invalidRequest(
|
||||
`${route} requires inline media (bytes or base64); ${asset.source.type} sources must be materialized first`,
|
||||
)
|
||||
|
||||
/** The remote URL of a `url` asset, for protocols that accept `http(s)` references natively. */
|
||||
export const mediaUrl = (asset: Media.Asset) => (asset.source.type === "url" ? asset.source.url : undefined)
|
||||
|
||||
export type MediaReference = { readonly type: "dataUrl" | "url" | "ref"; readonly value: string }
|
||||
|
||||
/**
|
||||
* The one string a provider can address an asset by: inline payloads as a data URL, `url` sources as their URL, and
|
||||
* this provider's own `ref` as its id. Other providers' refs are never forwarded and fail typed; omit `provider` for
|
||||
* APIs with no file handles at all.
|
||||
*/
|
||||
export const mediaReference = (
|
||||
asset: Media.Asset,
|
||||
provider: ProviderID | undefined,
|
||||
label: string,
|
||||
): Effect.Effect<MediaReference, AIError> => {
|
||||
const inline = asset.inline()
|
||||
if (inline) return Effect.succeed({ type: "dataUrl", value: inline.dataUrl })
|
||||
const url = mediaUrl(asset)
|
||||
if (url) return Effect.succeed({ type: "url", value: url })
|
||||
if (provider !== undefined && asset.source.type === "ref" && asset.source.provider === provider)
|
||||
return Effect.succeed({ type: "ref", value: asset.source.id })
|
||||
const accepted = provider === undefined ? "" : `, and ${provider} references`
|
||||
return Effect.fail(invalidRequest(`${label} accepts inline bytes, data URLs, http(s) URLs${accepted}`))
|
||||
}
|
||||
|
||||
export const normalizeToolFile = (part: Tool.FileContent) =>
|
||||
normalizeMedia({ type: "media", mediaType: part.mime, data: part.uri, filename: part.name })
|
||||
/**
|
||||
* Lift a tool-result file into a `MediaPart`. Tool files carry either a data URL, an `http(s)` URL, or raw base64 in
|
||||
* `uri`; the declared `mime` wins over any data-URL prefix so tool authors control the type the model sees.
|
||||
*/
|
||||
export const toolFileMedia = (item: Tool.FileContent): MediaPart => {
|
||||
const parsed = Media.parseDataUrl(item.uri)
|
||||
const asset = parsed
|
||||
? Media.from({ ...parsed.source, mediaType: item.mime })
|
||||
: /^https?:\/\//.test(item.uri)
|
||||
? Media.url(item.uri, { mediaType: item.mime })
|
||||
: Media.base64(item.uri, item.mime)
|
||||
return Message.media(asset, { filename: item.name })
|
||||
}
|
||||
|
||||
export const trimBaseUrl = (value: string) => value.replace(/\/+$/, "")
|
||||
|
||||
@@ -223,11 +253,11 @@ export const errorText = (error: unknown) => {
|
||||
|
||||
/**
|
||||
* `framing` step for Server-Sent Events. Decodes UTF-8, runs the SSE channel
|
||||
* decoder, optionally filters named events, and drops empty events. `[DONE]`
|
||||
* is dropped by default or retained for protocols that use it as their stream
|
||||
* boundary. Retry control events are ignored without interrupting the stream.
|
||||
* Decoder failures become provider output errors so the public error channel
|
||||
* stays `AIError`.
|
||||
* decoder, optionally filters named events, and drops empty and bare `null`
|
||||
* events. `[DONE]` is dropped by default or retained for protocols that use it
|
||||
* as their stream boundary. Retry control events are ignored without
|
||||
* interrupting the stream. Decoder failures become provider output errors so
|
||||
* the public error channel stays `AIError`.
|
||||
*/
|
||||
export const sseFraming = (
|
||||
bytes: Stream.Stream<Uint8Array, AIError>,
|
||||
@@ -257,6 +287,10 @@ export const sseFraming = (
|
||||
(event) =>
|
||||
(events === undefined || events.has(event.event)) &&
|
||||
event.data.length > 0 &&
|
||||
// Some OpenAI-compatible proxies serialize an empty flush as a bare
|
||||
// `data: null`, between events or after `[DONE]`. No protocol has a
|
||||
// null event, so it carries nothing and must not abort the stream.
|
||||
event.data !== "null" &&
|
||||
(event.data !== "[DONE]" || includeDone || (events !== undefined && event.event !== "message")),
|
||||
),
|
||||
Stream.map((event) => event.data),
|
||||
@@ -325,34 +359,6 @@ export const flattenToolRequest = (request: LLMRequest) => {
|
||||
}
|
||||
}
|
||||
|
||||
export const imageResponse = Effect.fn("ProviderShared.imageResponse")(function* (
|
||||
route: string,
|
||||
name: string,
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const http = new HttpContext({ url: response.request.url, status: response.status, headers: response.headers })
|
||||
const body = yield* response.text.pipe(
|
||||
Effect.mapError(
|
||||
(cause) =>
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({
|
||||
route,
|
||||
message: `Failed to read the ${name} response`,
|
||||
http,
|
||||
cause,
|
||||
}),
|
||||
}),
|
||||
),
|
||||
)
|
||||
return {
|
||||
body,
|
||||
invalid: (message: string, cause?: unknown) =>
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({ route, message, body, http, cause }),
|
||||
}),
|
||||
}
|
||||
})
|
||||
|
||||
export const matchToolChoice = <Auto, None, Required, Tool>(
|
||||
route: string,
|
||||
toolChoice: NonNullable<LLMRequest["toolChoice"]>,
|
||||
|
||||
@@ -77,7 +77,7 @@ function documentName(filename: string | undefined, names: Set<string>) {
|
||||
}
|
||||
|
||||
const mediaBase64 = Effect.fn("BedrockMedia.mediaBase64")(function* (part: MediaPart) {
|
||||
const media = ProviderShared.normalizeMedia(part)
|
||||
const media = yield* ProviderShared.requireInlineMedia("Bedrock Converse", part.media)
|
||||
const bytes = yield* Effect.fromResult(Encoding.decodeBase64(media.base64)).pipe(
|
||||
Effect.mapError((cause) =>
|
||||
ProviderShared.invalidRequest("Bedrock Converse media data must be valid base64", cause),
|
||||
@@ -92,13 +92,15 @@ const mediaBase64 = Effect.fn("BedrockMedia.mediaBase64")(function* (part: Media
|
||||
// get an image-specific error so the caller knows it's a format-support issue,
|
||||
// not a kind-detection issue.
|
||||
export const lower = Effect.fn("BedrockMedia.lower")(function* (part: MediaPart, documentNames: Set<string>) {
|
||||
const mime = part.mediaType.toLowerCase()
|
||||
const mime = part.media.mediaType.toLowerCase()
|
||||
const imageFormat = IMAGE_FORMATS[mime as keyof typeof IMAGE_FORMATS]
|
||||
if (imageFormat) {
|
||||
return [{ image: { format: imageFormat, source: { bytes: yield* mediaBase64(part) } } } satisfies ImageBlock]
|
||||
}
|
||||
if (mime.startsWith("image/"))
|
||||
return yield* ProviderShared.invalidRequest(`Bedrock Converse does not support image media type ${part.mediaType}`)
|
||||
return yield* ProviderShared.invalidRequest(
|
||||
`Bedrock Converse does not support image media type ${part.media.mediaType}`,
|
||||
)
|
||||
const documentFormat = DOCUMENT_FORMATS[mime as keyof typeof DOCUMENT_FORMATS]
|
||||
if (documentFormat) {
|
||||
const name = documentName(part.filename, documentNames)
|
||||
@@ -112,7 +114,7 @@ export const lower = Effect.fn("BedrockMedia.lower")(function* (part: MediaPart,
|
||||
]
|
||||
: [block]
|
||||
}
|
||||
return yield* ProviderShared.invalidRequest(`Bedrock Converse does not support media type ${part.mediaType}`)
|
||||
return yield* ProviderShared.invalidRequest(`Bedrock Converse does not support media type ${part.media.mediaType}`)
|
||||
})
|
||||
|
||||
export * as BedrockMedia from "./bedrock-media.js"
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
import { Effect, Schema, type Stream } from "effect"
|
||||
import type { Media } from "../../media.js"
|
||||
import { Framing } from "../../route/framing.js"
|
||||
import type { MediaProtocol } from "../../route/media-protocol.js"
|
||||
import { AIError, ContentPolicyError, ProviderID, type MediaUsage, type ProviderMetadata } from "../../schema/index.js"
|
||||
import { ProviderShared } from "../shared.js"
|
||||
import { MediaInput } from "./media-input.js"
|
||||
|
||||
const PROVIDER = ProviderID.make("google")
|
||||
|
||||
const UsageMetadata = Schema.Struct({
|
||||
promptTokenCount: Schema.optional(Schema.Number),
|
||||
candidatesTokenCount: Schema.optional(Schema.Number),
|
||||
totalTokenCount: Schema.optional(Schema.Number),
|
||||
})
|
||||
type UsageMetadata = Schema.Schema.Type<typeof UsageMetadata>
|
||||
|
||||
export const chunk = <const Part extends Schema.Top>(part: Part) =>
|
||||
Schema.Struct({
|
||||
candidates: Schema.optional(
|
||||
Schema.Array(
|
||||
Schema.Struct({
|
||||
content: Schema.optional(Schema.Struct({ parts: Schema.optional(Schema.Array(part)) })),
|
||||
finishReason: Schema.optional(Schema.String),
|
||||
}),
|
||||
),
|
||||
),
|
||||
promptFeedback: Schema.optional(
|
||||
Schema.Struct({
|
||||
blockReason: Schema.optional(Schema.String),
|
||||
blockReasonMessage: Schema.optional(Schema.String),
|
||||
}),
|
||||
),
|
||||
usageMetadata: Schema.optional(UsageMetadata),
|
||||
modelVersion: Schema.optional(Schema.String),
|
||||
responseId: Schema.optional(Schema.String),
|
||||
})
|
||||
|
||||
interface Chunk {
|
||||
readonly candidates?: ReadonlyArray<{ readonly finishReason?: string }>
|
||||
readonly promptFeedback?: { readonly blockReason?: string; readonly blockReasonMessage?: string }
|
||||
readonly usageMetadata?: UsageMetadata
|
||||
readonly modelVersion?: string
|
||||
readonly responseId?: string
|
||||
}
|
||||
|
||||
export interface Metadata {
|
||||
readonly usage?: UsageMetadata
|
||||
readonly finishReason?: string
|
||||
readonly modelVersion?: string
|
||||
readonly responseId?: string
|
||||
}
|
||||
|
||||
export const track = <State extends Metadata>(state: State, chunk: Chunk): State => ({
|
||||
...state,
|
||||
usage: chunk.usageMetadata ?? state.usage,
|
||||
finishReason: chunk.candidates?.[0]?.finishReason ?? state.finishReason,
|
||||
modelVersion: chunk.modelVersion ?? state.modelVersion,
|
||||
responseId: chunk.responseId ?? state.responseId,
|
||||
})
|
||||
|
||||
export const blocked = (name: string, chunk: Chunk, frame: string) => {
|
||||
const feedback = chunk.promptFeedback
|
||||
if (feedback?.blockReason === undefined) return undefined
|
||||
return new AIError({
|
||||
reason: new ContentPolicyError({
|
||||
message: `${name} blocked the request (${feedback.blockReason})${
|
||||
feedback.blockReasonMessage === undefined ? "" : `: ${feedback.blockReasonMessage}`
|
||||
}`,
|
||||
body: frame,
|
||||
}),
|
||||
})
|
||||
}
|
||||
|
||||
export const usage = (usage: UsageMetadata | undefined): MediaUsage | undefined =>
|
||||
usage === undefined
|
||||
? undefined
|
||||
: {
|
||||
type: "tokens",
|
||||
input: usage.promptTokenCount,
|
||||
output: usage.candidatesTokenCount,
|
||||
total: ProviderShared.totalTokens(usage.promptTokenCount, usage.candidatesTokenCount, usage.totalTokenCount),
|
||||
details: { google: usage },
|
||||
}
|
||||
|
||||
export const providerMetadata = (state: Metadata): ProviderMetadata => ({
|
||||
google: { finishReason: state.finishReason, modelVersion: state.modelVersion, responseId: state.responseId },
|
||||
})
|
||||
|
||||
export const path = (model: string, mode: MediaProtocol.Mode) =>
|
||||
mode === "stream" ? `/models/${model}:streamGenerateContent?alt=sse` : `/models/${model}:generateContent`
|
||||
|
||||
// `generateContent` answers with one document shaped exactly like a streamed chunk, so it is a single frame.
|
||||
export const frames = (bytes: Stream.Stream<Uint8Array, AIError>, mode: MediaProtocol.Mode) =>
|
||||
mode === "stream" ? Framing.sse.frame(bytes) : Framing.document.frame(bytes)
|
||||
|
||||
// Gemini does not fetch public URLs; inline payloads and Gemini Files references are the accepted inputs.
|
||||
export const mediaPart = (
|
||||
route: string,
|
||||
asset: Media.Asset,
|
||||
): Effect.Effect<
|
||||
| { readonly fileData: { readonly mimeType: string; readonly fileUri: string } }
|
||||
| { readonly inlineData: { readonly mimeType: string; readonly data: string } },
|
||||
AIError
|
||||
> => {
|
||||
const fileUri = MediaInput.refID(asset, PROVIDER)
|
||||
if (fileUri !== undefined) return Effect.succeed({ fileData: { mimeType: asset.mediaType, fileUri } })
|
||||
return ProviderShared.requireInlineMedia(route, asset).pipe(
|
||||
Effect.map((media) => ({ inlineData: { mimeType: media.mime, data: media.base64 } })),
|
||||
)
|
||||
}
|
||||
|
||||
export * as GeminiGenerateContent from "./gemini-generate-content.js"
|
||||
@@ -1,31 +0,0 @@
|
||||
import { Effect, Encoding } from "effect"
|
||||
import type { ImageInput } from "../../image.js"
|
||||
import { InvalidRequestError, AIError } from "../../schema/index.js"
|
||||
|
||||
const invalid = (message: string, cause?: unknown) =>
|
||||
new AIError({
|
||||
reason: new InvalidRequestError({ message, cause }),
|
||||
})
|
||||
|
||||
export const dataUrl = (input: Extract<ImageInput, { readonly type: "bytes" }>) =>
|
||||
`data:${input.mediaType};base64,${Encoding.encodeBase64(input.data)}`
|
||||
|
||||
export const decodeDataUrl = (
|
||||
url: string,
|
||||
): Effect.Effect<{ readonly mediaType: string; readonly data: Uint8Array } | undefined, AIError> => {
|
||||
if (!url.startsWith("data:")) return Effect.undefined
|
||||
const match = /^data:([^;,]+);base64,(.*)$/s.exec(url)
|
||||
if (!match) return Effect.fail(invalid("Image data URLs must contain a MIME type and base64 data"))
|
||||
return Effect.fromResult(Encoding.decodeBase64(match[2])).pipe(
|
||||
Effect.mapError((cause) => invalid("Image data URL contains invalid base64 data", cause)),
|
||||
Effect.map((data) => ({ mediaType: match[1], data })),
|
||||
)
|
||||
}
|
||||
|
||||
export const invalidImageInput = invalid
|
||||
|
||||
export const ImageInputs = {
|
||||
dataUrl,
|
||||
decodeDataUrl,
|
||||
invalid: invalidImageInput,
|
||||
} as const
|
||||
@@ -0,0 +1,54 @@
|
||||
import { Effect, Encoding } from "effect"
|
||||
import { Media } from "../../media.js"
|
||||
import type { MediaProtocol } from "../../route/media-protocol.js"
|
||||
import type { AIError, ProviderID } from "../../schema/index.js"
|
||||
import { ProviderShared } from "../shared.js"
|
||||
|
||||
/** Owned bytes for multipart uploads; decodes `base64` sources and rejects remote sources. */
|
||||
export const inlineBytes = (route: string, asset: Media.Asset): Effect.Effect<Uint8Array, AIError> => {
|
||||
if (asset.source.type === "bytes") return Effect.succeed(asset.source.data)
|
||||
const inline = asset.inline()
|
||||
if (!inline) return Effect.fail(ProviderShared.inlineRequired(route, asset))
|
||||
return Effect.fromResult(Encoding.decodeBase64(inline.base64)).pipe(
|
||||
Effect.mapError((cause) => ProviderShared.invalidRequest(`${route} media contains invalid base64 data`, cause)),
|
||||
)
|
||||
}
|
||||
|
||||
/** Copied because `BlobPart` requires a plain `ArrayBuffer`. */
|
||||
export const blob = (data: Uint8Array, mediaType: string) => {
|
||||
const buffer = new ArrayBuffer(data.byteLength)
|
||||
new Uint8Array(buffer).set(data)
|
||||
return new Blob([buffer], { type: mediaType })
|
||||
}
|
||||
|
||||
const isScalar = (value: unknown): value is string | number | boolean =>
|
||||
typeof value === "string" || typeof value === "number" || typeof value === "boolean"
|
||||
|
||||
export const query = (route: string, values: Record<string, unknown>): Effect.Effect<MediaProtocol.Query, AIError> => {
|
||||
const entries = Object.entries(values).filter(([, value]) => value !== undefined)
|
||||
const invalid = entries.find(([, value]) => !isScalar(value) && !(Array.isArray(value) && value.every(isScalar)))
|
||||
if (invalid !== undefined)
|
||||
return Effect.fail(ProviderShared.invalidRequest(`${route} cannot send "${invalid[0]}" as a query parameter`))
|
||||
return Effect.succeed(
|
||||
Object.fromEntries(entries.map(([key, value]) => [key, Array.isArray(value) ? value.map(String) : String(value)])),
|
||||
)
|
||||
}
|
||||
|
||||
/** Provider file handle when the ref belongs to this provider; refs from other providers are never forwarded. */
|
||||
export const refID = (asset: Media.Asset, provider: ProviderID) =>
|
||||
asset.source.type === "ref" && asset.source.provider === provider ? asset.source.id : undefined
|
||||
|
||||
/** Decode a provider's base64 output once into an owned `bytes` asset, sniffing the type when it is not declared. */
|
||||
export const decodedAsset = (
|
||||
invalid: (message: string, cause?: unknown) => AIError,
|
||||
label: string,
|
||||
data: string,
|
||||
mediaType: string | undefined,
|
||||
options?: Media.AssetOptions,
|
||||
) =>
|
||||
Effect.fromResult(Encoding.decodeBase64(data)).pipe(
|
||||
Effect.mapError((cause) => invalid(`${label} contains invalid base64 data`, cause)),
|
||||
Effect.map((bytes) => Media.bytes(bytes, mediaType, options)),
|
||||
)
|
||||
|
||||
export * as MediaInput from "./media-input.js"
|
||||
@@ -17,6 +17,7 @@ import { RequestExecutor } from "../../route/executor.js"
|
||||
import { HttpTransport } from "../../route/transport/index.js"
|
||||
import { OpenResponses } from "../open-responses.js"
|
||||
import { JsonObject, optionalNull, ProviderShared } from "../shared.js"
|
||||
import { Media } from "../../media.js"
|
||||
|
||||
const Body = Schema.Struct({
|
||||
model: Schema.String,
|
||||
@@ -157,20 +158,22 @@ function toMessage(item: (typeof Response.Type.output)[number], model: LLMReques
|
||||
if (part.type === "input_image")
|
||||
return {
|
||||
type: "media",
|
||||
data: part.image_url,
|
||||
mediaType: /^data:([^;,]+)/.exec(part.image_url)?.[1] ?? "image/*",
|
||||
media: replayMedia(part.image_url, "image/*"),
|
||||
providerMetadata: part.detail === undefined ? undefined : { [key]: { detail: part.detail } },
|
||||
}
|
||||
const data = part.file_url === undefined ? part.file_data : part.file_url
|
||||
return {
|
||||
type: "media",
|
||||
data,
|
||||
media: replayMedia(part.file_url === undefined ? part.file_data : part.file_url, "application/octet-stream"),
|
||||
filename: part.filename,
|
||||
mediaType: /^data:([^;,]+)/.exec(data)?.[1] ?? "application/octet-stream",
|
||||
providerMetadata: part.detail === undefined ? undefined : { [key]: { detail: part.detail } },
|
||||
}
|
||||
}),
|
||||
})
|
||||
}
|
||||
|
||||
/** Replayed compaction items carry either a data URL or a remote URL; the data URL's own type wins when present. */
|
||||
const replayMedia = (value: string, fallbackType: string) =>
|
||||
Media.parseDataUrl(value) ??
|
||||
(/^https?:\/\//.test(value) ? Media.url(value, { mediaType: fallbackType }) : Media.base64(value, fallbackType))
|
||||
|
||||
export * as ResponsesCompaction from "./responses-compaction.js"
|
||||
|
||||
@@ -0,0 +1,118 @@
|
||||
import { Effect } from "effect"
|
||||
import { Media } from "../../media.js"
|
||||
import { MediaProtocol } from "../../route/media-protocol.js"
|
||||
import type { AIError, MediaUsage, ProviderID, ProviderMetadata } from "../../schema/index.js"
|
||||
import {
|
||||
SpeechAudioDeltaEvent,
|
||||
SpeechFinishEvent,
|
||||
SpeechTimestampsEvent,
|
||||
type SpeechEvent,
|
||||
type SpeechVoice,
|
||||
} from "../../speech.js"
|
||||
import { concatBytes } from "../../utils/bytes.js"
|
||||
import { ProviderShared } from "../shared.js"
|
||||
|
||||
export interface Audio {
|
||||
/** Appended in place: the route creates fresh state for each response through `initial`. */
|
||||
readonly chunks: Array<Uint8Array>
|
||||
}
|
||||
|
||||
export type StepResult<State> = readonly [State, ReadonlyArray<SpeechEvent>]
|
||||
|
||||
/** Empty chunks (keep-alive records) emit nothing. */
|
||||
export const delta = <State extends Audio>(state: State, chunk: Uint8Array): StepResult<State> => {
|
||||
if (chunk.length === 0) return [state, []]
|
||||
state.chunks.push(chunk)
|
||||
return [state, [SpeechAudioDeltaEvent.make({ chunk })]]
|
||||
}
|
||||
|
||||
export const step =
|
||||
<State extends Audio>(onRecord: (state: State, frame: string) => Effect.Effect<StepResult<State>, AIError>) =>
|
||||
(state: State, frame: string | Uint8Array) =>
|
||||
typeof frame === "string" ? onRecord(state, frame) : Effect.succeed(delta(state, frame))
|
||||
|
||||
export const timestamps = (
|
||||
texts: ReadonlyArray<string>,
|
||||
starts: ReadonlyArray<number>,
|
||||
ends: ReadonlyArray<number>,
|
||||
): ReadonlyArray<SpeechEvent> =>
|
||||
texts.length === 0
|
||||
? []
|
||||
: [
|
||||
SpeechTimestampsEvent.make({
|
||||
items: texts.map((text, index) => ({ text, startSeconds: starts[index] ?? 0, endSeconds: ends[index] ?? 0 })),
|
||||
}),
|
||||
]
|
||||
|
||||
export const voiceID = (voice: SpeechVoice | undefined) => (typeof voice === "object" ? voice.id : voice)
|
||||
|
||||
const CONTAINER_MEDIA_TYPES: Readonly<Record<string, string>> = {
|
||||
mp3: "audio/mpeg",
|
||||
wav: "audio/wav",
|
||||
opus: "audio/ogg",
|
||||
aac: "audio/aac",
|
||||
flac: "audio/flac",
|
||||
}
|
||||
|
||||
export const container = (format: string, sampleRate?: number) => ({
|
||||
mediaType: CONTAINER_MEDIA_TYPES[format],
|
||||
info: { format, sampleRate },
|
||||
})
|
||||
|
||||
const PCM_MEDIA_TYPES = {
|
||||
pcm_s16le: "audio/pcm",
|
||||
pcm_f32le: "audio/pcm",
|
||||
pcm_mulaw: "audio/mulaw",
|
||||
pcm_alaw: "audio/alaw",
|
||||
} as const
|
||||
|
||||
export type PcmEncoding = keyof typeof PCM_MEDIA_TYPES
|
||||
|
||||
export const pcm = (encoding: PcmEncoding, sampleRate: number | undefined, mediaType?: string) => ({
|
||||
mediaType: mediaType ?? PCM_MEDIA_TYPES[encoding],
|
||||
info: { format: "pcm", encoding, sampleRate, channels: 1 },
|
||||
})
|
||||
|
||||
export const sampleRate = (mediaType: string | undefined) => {
|
||||
const rate = /rate=(\d+)/i.exec(mediaType ?? "")?.[1]
|
||||
return rate === undefined ? undefined : Number(rate)
|
||||
}
|
||||
|
||||
export const unsupportedFormat = (provider: ProviderID, route: string, message: string) =>
|
||||
ProviderShared.unsupportedOperation({ operation: "media.format", provider, route, message })
|
||||
|
||||
/** A declared `mediaType` wins over sniffing: headerless PCM can start with bytes that look like an MPEG frame sync. */
|
||||
export const finish = (
|
||||
route: string,
|
||||
state: Audio,
|
||||
output: {
|
||||
readonly mediaType: string | undefined
|
||||
readonly info?: Media.Info
|
||||
readonly usage?: MediaUsage
|
||||
readonly providerMetadata?: ProviderMetadata
|
||||
readonly detail?: string
|
||||
},
|
||||
): Effect.Effect<ReadonlyArray<SpeechEvent>, AIError> => {
|
||||
if (state.chunks.length === 0)
|
||||
return Effect.fail(
|
||||
MediaProtocol.frameError(
|
||||
route,
|
||||
`The provider returned no audio${output.detail === undefined ? "" : ` (${output.detail})`}`,
|
||||
),
|
||||
)
|
||||
return Effect.succeed([
|
||||
SpeechFinishEvent.make({
|
||||
audio: Media.bytes(concatBytes(state.chunks), output.mediaType, { info: output.info }),
|
||||
usage: output.usage,
|
||||
providerMetadata: output.providerMetadata,
|
||||
}),
|
||||
])
|
||||
}
|
||||
|
||||
export const headerUsage = (type: "characters" | "credits", value: string | undefined): MediaUsage | undefined => {
|
||||
const amount = Number(value)
|
||||
if (!Number.isFinite(amount)) return undefined
|
||||
return type === "credits" ? { type, credits: amount } : { type, characters: amount }
|
||||
}
|
||||
|
||||
export * as SpeechStream from "./speech-stream.js"
|
||||
@@ -1,61 +1,38 @@
|
||||
import { Effect, Encoding, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import { GeneratedImage, ImageModel, ImageResponse, type ImageRequestFor, type ImageRoute } from "../image.js"
|
||||
import { Auth, type Definition as AuthDefinition } from "../route/auth.js"
|
||||
import { Usage, mergeHttpOptions, mergeJsonRecords, type HttpOptions } from "../schema/index.js"
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { ProviderShared, optionalNull } from "./shared.js"
|
||||
import { ImageInputs } from "./utils/image-input.js"
|
||||
import { MediaInput } from "./utils/media-input.js"
|
||||
|
||||
const ADAPTER = "xai-images"
|
||||
const NAME = "xAI Images"
|
||||
const PROVIDER = ProviderID.make("xai")
|
||||
export const DEFAULT_BASE_URL = "https://api.x.ai/v1"
|
||||
export const PATH = "/images/generations"
|
||||
export const EDIT_PATH = "/images/edits"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type XAIImageString<Known extends string> = Known | (string & {})
|
||||
|
||||
/** Provider-native options. Common fields (`n`, `aspectRatio`, `images`) live on the request. */
|
||||
export type XAIImageOptions = {
|
||||
readonly n?: number
|
||||
readonly aspectRatio?: XAIImageString<
|
||||
| "1:1"
|
||||
| "3:4"
|
||||
| "4:3"
|
||||
| "9:16"
|
||||
| "16:9"
|
||||
| "2:3"
|
||||
| "3:2"
|
||||
| "9:19.5"
|
||||
| "19.5:9"
|
||||
| "9:20"
|
||||
| "20:9"
|
||||
| "1:2"
|
||||
| "2:1"
|
||||
| "auto"
|
||||
>
|
||||
readonly aspect_ratio?: XAIImageString<
|
||||
| "1:1"
|
||||
| "3:4"
|
||||
| "4:3"
|
||||
| "9:16"
|
||||
| "16:9"
|
||||
| "2:3"
|
||||
| "3:2"
|
||||
| "9:19.5"
|
||||
| "19.5:9"
|
||||
| "9:20"
|
||||
| "20:9"
|
||||
| "1:2"
|
||||
| "2:1"
|
||||
| "auto"
|
||||
>
|
||||
readonly resolution?: XAIImageString<"1k" | "2k">
|
||||
readonly responseFormat?: XAIImageString<"url" | "b64_json">
|
||||
readonly response_format?: XAIImageString<"url" | "b64_json">
|
||||
} & Record<string, unknown>
|
||||
|
||||
type XAIImageBody = Record<string, unknown> & {
|
||||
readonly model: string
|
||||
readonly prompt: string
|
||||
}
|
||||
export type Request = ImageRequestFor<XAIImageOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Response schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const XAIImageResponse = Schema.Struct({
|
||||
data: Schema.Array(
|
||||
@@ -69,120 +46,106 @@ const XAIImageResponse = Schema.Struct({
|
||||
usage: Schema.optional(Schema.Unknown),
|
||||
})
|
||||
|
||||
export interface ModelInput {
|
||||
readonly id: string
|
||||
readonly auth: AuthDefinition
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const nativeOptions = (options: XAIImageOptions | undefined) => {
|
||||
if (!options) return undefined
|
||||
const { aspectRatio, responseFormat, ...native } = options
|
||||
return {
|
||||
aspect_ratio: aspectRatio,
|
||||
response_format: responseFormat,
|
||||
...native,
|
||||
}
|
||||
const { responseFormat, ...native } = options
|
||||
return { response_format: responseFormat, ...native }
|
||||
}
|
||||
|
||||
const applyQuery = (url: string, query: Record<string, string> | undefined) => {
|
||||
if (!query) return url
|
||||
const next = new URL(url)
|
||||
Object.entries(query).forEach(([key, value]) => next.searchParams.set(key, value))
|
||||
return next.toString()
|
||||
}
|
||||
const isEdit = (request: Request) => (request.images?.length ?? 0) > 0
|
||||
|
||||
export const model = (input: ModelInput) => {
|
||||
const route: ImageRoute<XAIImageOptions> = {
|
||||
id: ADAPTER,
|
||||
generate: Effect.fn("XAIImages.generate")(function* (request: ImageRequestFor<XAIImageOptions>, execute) {
|
||||
const http = mergeHttpOptions(request.model.http, request.http)
|
||||
const imageReferences = (request.images ?? []).map((image) => {
|
||||
if (image.type === "bytes") return { url: ImageInputs.dataUrl(image), type: "image_url" as const }
|
||||
if (image.type === "url") return { url: image.url, type: "image_url" as const }
|
||||
if (image.type === "file-id") return { file_id: image.id }
|
||||
return undefined
|
||||
})
|
||||
if (imageReferences.some((image) => image === undefined))
|
||||
return yield* ImageInputs.invalid("xAI Images accepts image URLs, data URLs, bytes, and file IDs")
|
||||
const requestBody = mergeJsonRecords(
|
||||
const reference = (asset: Media.Asset) =>
|
||||
ProviderShared.mediaReference(asset, PROVIDER, NAME).pipe(
|
||||
Effect.map((item) =>
|
||||
item.type === "ref" ? { file_id: item.value } : { url: item.value, type: "image_url" as const },
|
||||
),
|
||||
)
|
||||
|
||||
const fromRequest = Effect.fn("XAIImages.fromRequest")(function* (request: Request) {
|
||||
const references = yield* Effect.forEach(request.images ?? [], reference)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
// xAI takes one edit source as `image` and several as `images`.
|
||||
image: references.length === 1 ? references[0] : undefined,
|
||||
images: references.length > 1 ? references : undefined,
|
||||
n: request.n,
|
||||
aspect_ratio: request.aspectRatio,
|
||||
},
|
||||
nativeOptions(request.providerOptions),
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeResponse = Effect.fn("XAIImages.decodeResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, XAIImageResponse)(response)
|
||||
const decoded = output.value
|
||||
const images = yield* Effect.forEach(decoded.data, (item, index) => {
|
||||
const providerMetadata =
|
||||
item.revised_prompt === undefined || item.revised_prompt === null
|
||||
? undefined
|
||||
: { xai: { revisedPrompt: item.revised_prompt } }
|
||||
if (item.b64_json)
|
||||
return MediaInput.decodedAsset(
|
||||
output.invalid,
|
||||
`${NAME} result ${index}`,
|
||||
item.b64_json,
|
||||
item.mime_type ?? undefined,
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
image: imageReferences.length === 1 ? imageReferences[0] : undefined,
|
||||
images: imageReferences.length > 1 ? imageReferences : undefined,
|
||||
providerMetadata,
|
||||
},
|
||||
nativeOptions(request.options),
|
||||
http?.body,
|
||||
) as XAIImageBody
|
||||
const text = ProviderShared.encodeJson(requestBody)
|
||||
const url = applyQuery(
|
||||
`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${imageReferences.length === 0 ? PATH : EDIT_PATH}`,
|
||||
http?.query,
|
||||
)
|
||||
const headers = yield* Auth.toEffect(input.auth)({
|
||||
request,
|
||||
method: "POST",
|
||||
url,
|
||||
body: text,
|
||||
headers: Headers.fromInput({ ...input.headers, ...http?.headers }),
|
||||
})
|
||||
const response = yield* execute(
|
||||
HttpClientRequest.post(url).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(text, "application/json"),
|
||||
),
|
||||
)
|
||||
const output = yield* ProviderShared.imageResponse(ADAPTER, "xAI Images", response)
|
||||
const decoded = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(XAIImageResponse))(output.body).pipe(
|
||||
Effect.mapError((cause) => output.invalid("xAI Images returned an invalid response", cause)),
|
||||
)
|
||||
const images = yield* Effect.forEach(decoded.data, (item, index) => {
|
||||
const mediaType = item.mime_type ?? "application/octet-stream"
|
||||
if (item.b64_json)
|
||||
return Effect.fromResult(Encoding.decodeBase64(item.b64_json)).pipe(
|
||||
Effect.mapError((cause) =>
|
||||
output.invalid(`xAI Images result ${index} contains invalid base64 data`, cause),
|
||||
),
|
||||
Effect.map(
|
||||
(data) =>
|
||||
new GeneratedImage({
|
||||
mediaType,
|
||||
data,
|
||||
providerMetadata:
|
||||
item.revised_prompt === undefined || item.revised_prompt === null
|
||||
? undefined
|
||||
: { xai: { revisedPrompt: item.revised_prompt } },
|
||||
}),
|
||||
),
|
||||
)
|
||||
if (item.url)
|
||||
return Effect.succeed(
|
||||
new GeneratedImage({
|
||||
mediaType,
|
||||
data: item.url,
|
||||
providerMetadata:
|
||||
item.revised_prompt === undefined || item.revised_prompt === null
|
||||
? undefined
|
||||
: { xai: { revisedPrompt: item.revised_prompt } },
|
||||
}),
|
||||
)
|
||||
return Effect.fail(output.invalid(`xAI Images result ${index} has neither image data nor a URL`))
|
||||
})
|
||||
if (images.length === 0) return yield* output.invalid("xAI Images returned no images")
|
||||
const usage = ProviderShared.isRecord(decoded.usage) ? decoded.usage : undefined
|
||||
return new ImageResponse({
|
||||
images,
|
||||
usage: usage === undefined ? undefined : new Usage({ providerMetadata: { xai: usage } }),
|
||||
providerMetadata: usage === undefined ? undefined : { xai: { usage } },
|
||||
})
|
||||
}),
|
||||
}
|
||||
return ImageModel.make<XAIImageOptions>({ id: input.id, provider: "xai", route, http: input.http })
|
||||
}
|
||||
if (item.url)
|
||||
return Effect.succeed(Media.url(item.url, { mediaType: item.mime_type ?? undefined, providerMetadata }))
|
||||
return Effect.fail(output.invalid(`${NAME} result ${index} has neither image data nor a URL`))
|
||||
})
|
||||
if (images.length === 0) return yield* output.invalid(`${NAME} returned no images`)
|
||||
const usage = ProviderShared.isRecord(decoded.usage) ? decoded.usage : undefined
|
||||
// xAI reports image counts rather than tokens, seconds, or credits; the raw record stays in provider metadata.
|
||||
return new ImageResponse({
|
||||
images,
|
||||
providerMetadata: usage === undefined ? undefined : { xai: { usage } },
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.inline<Request, ImageResponse>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["mask", "size", "seed", "format"],
|
||||
body: { from: fromRequest },
|
||||
response: { decode: decodeResponse },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
ImageModel.fromRoute<XAIImageOptions>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => (isEdit(request) ? EDIT_PATH : PATH),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const XAIImages = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
|
||||
@@ -0,0 +1,211 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import type { Status } from "../generation.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
import { VideoModel, VideoResponse, type VideoRequestFor } from "../video.js"
|
||||
import { ProviderShared, optionalNull } from "./shared.js"
|
||||
|
||||
const ADAPTER = "xai-video"
|
||||
const NAME = "xAI Video"
|
||||
const PROVIDER = ProviderID.make("xai")
|
||||
export const DEFAULT_BASE_URL = "https://api.x.ai/v1"
|
||||
export const PATH = "/videos/generations"
|
||||
export const EDIT_PATH = "/videos/edits"
|
||||
export const EXTEND_PATH = "/videos/extensions"
|
||||
export const STATUS_PATH = "/videos"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Provider-native options. Common fields (`frames`, `references`, `video`, `durationSeconds`, `aspectRatio`,
|
||||
* `resolution`, `audio`) live on the request. `mode` selects the endpoint a `video` input is sent to.
|
||||
*/
|
||||
export type XAIVideoOptions = {
|
||||
readonly mode?: "edit" | "extend"
|
||||
readonly reference_audios?: ReadonlyArray<{ readonly voice_id: string }>
|
||||
} & Record<string, unknown>
|
||||
|
||||
export type Request = VideoRequestFor<XAIVideoOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Token and response schemas
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const Token = Schema.Struct({ requestID: Schema.String })
|
||||
export type Token = Schema.Schema.Type<typeof Token>
|
||||
|
||||
const StartResponse = Schema.Struct({ request_id: Schema.String })
|
||||
|
||||
const VideoStatus = Schema.Struct({
|
||||
status: Schema.String,
|
||||
progress: optionalNull(Schema.Number),
|
||||
video: optionalNull(
|
||||
Schema.Struct({
|
||||
url: optionalNull(Schema.String),
|
||||
duration: optionalNull(Schema.Number),
|
||||
respect_moderation: optionalNull(Schema.Boolean),
|
||||
}),
|
||||
),
|
||||
error: optionalNull(
|
||||
Schema.Struct({
|
||||
code: optionalNull(Schema.String),
|
||||
message: optionalNull(Schema.String),
|
||||
}),
|
||||
),
|
||||
model: optionalNull(Schema.String),
|
||||
})
|
||||
|
||||
const STATUS = {
|
||||
pending: "running",
|
||||
done: "completed",
|
||||
failed: "failed",
|
||||
expired: "expired",
|
||||
} as const satisfies Record<string, Status>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const mediaInput = (asset: Media.Asset) =>
|
||||
ProviderShared.mediaReference(asset, PROVIDER, NAME).pipe(
|
||||
Effect.map((reference) => (reference.type === "ref" ? { file_id: reference.value } : { url: reference.value })),
|
||||
)
|
||||
|
||||
const nativeOptions = (options: XAIVideoOptions | undefined) => {
|
||||
if (!options) return undefined
|
||||
const { mode: _mode, ...native } = options
|
||||
return native
|
||||
}
|
||||
|
||||
const fromRequest = Effect.fn("XAIVideo.fromRequest")(function* (request: Request) {
|
||||
const image = request.frames?.first === undefined ? undefined : yield* mediaInput(request.frames.first)
|
||||
const lastFrame = request.frames?.last === undefined ? undefined : yield* mediaInput(request.frames.last)
|
||||
const video = request.video === undefined ? undefined : yield* mediaInput(request.video)
|
||||
const references = yield* Effect.forEach(request.references ?? [], mediaInput)
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{
|
||||
model: request.model.id,
|
||||
prompt: request.prompt,
|
||||
image,
|
||||
last_frame: lastFrame,
|
||||
reference_images: references.length === 0 ? undefined : references,
|
||||
video,
|
||||
duration: request.durationSeconds,
|
||||
aspect_ratio: request.aspectRatio,
|
||||
resolution: request.resolution,
|
||||
generate_audio: request.audio,
|
||||
},
|
||||
nativeOptions(request.providerOptions),
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeStart = MediaProtocol.decodeStarted(ADAPTER, NAME, StartResponse, (value) => ({
|
||||
token: { requestID: value.request_id },
|
||||
snapshot: { id: value.request_id, status: "running" },
|
||||
}))
|
||||
|
||||
// `progress` is undocumented but observed live as a 0..100 percentage (recorded cassette: 1 → 10 → 37 → 100).
|
||||
const fraction = (progress: number | null | undefined) =>
|
||||
progress !== undefined && progress !== null && progress >= 0 && progress <= 100 ? progress / 100 : undefined
|
||||
|
||||
const decodeVideoStatus = MediaProtocol.decodeJson(ADAPTER, NAME, VideoStatus)
|
||||
|
||||
const decodeStatus = Effect.fn("XAIVideo.decodeStatus")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeVideoStatus(response)
|
||||
const status = yield* MediaProtocol.status(STATUS, output.value.status, output)
|
||||
return { id: context.token.requestID, status, progress: fraction(output.value.progress) }
|
||||
})
|
||||
|
||||
const decodeResult = Effect.fn("XAIVideo.decodeResult")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) {
|
||||
const output = yield* decodeVideoStatus(response)
|
||||
const decoded = output.value
|
||||
const status = yield* MediaProtocol.status(STATUS, decoded.status, output)
|
||||
if (status === "running") return yield* output.invalid(`${NAME} request ${context.token.requestID} has not finished`)
|
||||
if (status === "failed") {
|
||||
const code = decoded.error?.code ?? undefined
|
||||
const message = decoded.error?.message ?? undefined
|
||||
return yield* output.ended(
|
||||
"failed",
|
||||
`${NAME} generation failed${code === undefined ? "" : ` (${code})`}${message === undefined ? "" : `: ${message}`}`,
|
||||
)
|
||||
}
|
||||
if (status !== "completed")
|
||||
return yield* output.ended("expired", `${NAME} request ${context.token.requestID} expired`)
|
||||
// `respect_moderation: false` marks a filtered result; a URL may still be present, so report it as a notice.
|
||||
const notices =
|
||||
decoded.video?.respect_moderation === false
|
||||
? [{ type: "moderated" as const, message: `${NAME} flagged the generated video for moderation` }]
|
||||
: undefined
|
||||
const url = decoded.video?.url ?? undefined
|
||||
if (url === undefined && notices !== undefined)
|
||||
return yield* output.contentPolicy(`${NAME} withheld the video for moderation`)
|
||||
if (url === undefined) return yield* output.invalid(`${NAME} completed without a video URL`)
|
||||
const duration = decoded.video?.duration ?? undefined
|
||||
return new VideoResponse({
|
||||
videos: [
|
||||
Media.url(url, {
|
||||
mediaType: "video/mp4",
|
||||
info: duration === undefined ? undefined : { durationSeconds: duration },
|
||||
}),
|
||||
],
|
||||
notices,
|
||||
providerMetadata: { xai: { requestId: context.token.requestID, model: decoded.model ?? undefined } },
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const statusPath = (token: Token) => `${STATUS_PATH}/${token.requestID}`
|
||||
|
||||
export const protocol = MediaProtocol.queued<Request, VideoResponse, Token>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
token: Token,
|
||||
unsupported: ["n", "seed", "negativePrompt"],
|
||||
start: { body: { from: fromRequest }, decode: decodeStart },
|
||||
status: { path: statusPath, decode: decodeStatus },
|
||||
result: { path: statusPath, decode: decodeResult },
|
||||
})
|
||||
|
||||
// A source video goes to `/videos/edits` unless `providerOptions.mode` asks for an extension.
|
||||
const startPath = (request: Request) => {
|
||||
if (request.video === undefined) return PATH
|
||||
return request.providerOptions?.mode === "extend" ? EXTEND_PATH : EDIT_PATH
|
||||
}
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
VideoModel.fromRoute<XAIVideoOptions, Token>(
|
||||
{
|
||||
id: ADAPTER,
|
||||
provider: PROVIDER,
|
||||
protocol,
|
||||
baseURL: DEFAULT_BASE_URL,
|
||||
path: ({ request }) => startPath(request),
|
||||
},
|
||||
input,
|
||||
)
|
||||
|
||||
export const XAIVideo = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
@@ -1,29 +1,34 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import { GeneratedImage, ImageModel, ImageResponse, type ImageRequestFor, type ImageRoute } from "../image.js"
|
||||
import { Auth, type Definition as AuthDefinition } from "../route/auth.js"
|
||||
import { mergeHttpOptions, mergeJsonRecords, type HttpOptions } from "../schema/index.js"
|
||||
import { ProviderShared } from "./shared.js"
|
||||
import { ImageInputs } from "./utils/image-input.js"
|
||||
import type { HttpClientResponse } from "effect/unstable/http"
|
||||
import { ImageModel, ImageResponse, type ImageRequestFor } from "../image.js"
|
||||
import { Media } from "../media.js"
|
||||
import { MediaProtocol } from "../route/media-protocol.js"
|
||||
import { MediaRoute } from "../route/media.js"
|
||||
import { ProviderID, mergeJsonRecords } from "../schema/index.js"
|
||||
|
||||
const ADAPTER = "zai-images"
|
||||
const NAME = "Z.ai Images"
|
||||
const PROVIDER = ProviderID.make("zai")
|
||||
export const DEFAULT_BASE_URL = "https://api.z.ai/api/paas/v4"
|
||||
export const PATH = "/images/generations"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 1. Public model input
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type ZAIImageString<Known extends string> = Known | (string & {})
|
||||
|
||||
/** Provider-native options. The common `size` field lives on the request. */
|
||||
export type ZAIImageOptions = {
|
||||
readonly size?: ZAIImageString<
|
||||
"1024x1024" | "768x1344" | "864x1152" | "1344x768" | "1152x864" | "1440x720" | "720x1440"
|
||||
>
|
||||
readonly quality?: ZAIImageString<"hd" | "standard">
|
||||
readonly userID?: string
|
||||
} & Record<string, unknown>
|
||||
|
||||
type ZAIImageBody = Record<string, unknown> & {
|
||||
readonly model: string
|
||||
readonly prompt: string
|
||||
}
|
||||
export type Request = ImageRequestFor<ZAIImageOptions>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 2. Response schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const ZAIImageResponse = Schema.Struct({
|
||||
created: Schema.optional(Schema.Int),
|
||||
@@ -40,84 +45,81 @@ const ZAIImageResponse = Schema.Struct({
|
||||
),
|
||||
})
|
||||
|
||||
export interface ModelInput {
|
||||
readonly id: string
|
||||
readonly auth: AuthDefinition
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// 5. Request body construction
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const nativeOptions = (options: ZAIImageOptions | undefined) => {
|
||||
if (!options) return undefined
|
||||
const { userID, ...native } = options
|
||||
return {
|
||||
user_id: userID,
|
||||
...native,
|
||||
}
|
||||
return { user_id: userID, ...native }
|
||||
}
|
||||
|
||||
const applyQuery = (url: string, query: Record<string, string> | undefined) => {
|
||||
if (!query) return url
|
||||
const next = new URL(url)
|
||||
Object.entries(query).forEach(([key, value]) => next.searchParams.set(key, value))
|
||||
return next.toString()
|
||||
}
|
||||
const fromRequest = Effect.fn("ZAIImages.fromRequest")(function* (request: Request) {
|
||||
return MediaProtocol.json(
|
||||
mergeJsonRecords(
|
||||
{ model: request.model.id, prompt: request.prompt, size: request.size },
|
||||
nativeOptions(request.providerOptions),
|
||||
request.http?.body,
|
||||
) ?? {},
|
||||
)
|
||||
})
|
||||
|
||||
export const model = (input: ModelInput) => {
|
||||
const route: ImageRoute<ZAIImageOptions> = {
|
||||
id: ADAPTER,
|
||||
generate: Effect.fn("ZAIImages.generate")(function* (request: ImageRequestFor<ZAIImageOptions>, execute) {
|
||||
if ((request.images?.length ?? 0) > 0)
|
||||
return yield* ImageInputs.invalid("Z.ai hosted image generation does not support image inputs")
|
||||
const http = mergeHttpOptions(request.model.http, request.http)
|
||||
const requestBody = mergeJsonRecords(
|
||||
{ model: request.model.id, prompt: request.prompt },
|
||||
nativeOptions(request.options),
|
||||
http?.body,
|
||||
) as ZAIImageBody
|
||||
const text = ProviderShared.encodeJson(requestBody)
|
||||
const url = applyQuery(`${(input.baseURL ?? DEFAULT_BASE_URL).replace(/\/$/, "")}${PATH}`, http?.query)
|
||||
const headers = yield* Auth.toEffect(input.auth)({
|
||||
request,
|
||||
method: "POST",
|
||||
url,
|
||||
body: text,
|
||||
headers: Headers.fromInput({ ...input.headers, ...http?.headers }),
|
||||
})
|
||||
const response = yield* execute(
|
||||
HttpClientRequest.post(url).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(text, "application/json"),
|
||||
),
|
||||
)
|
||||
const output = yield* ProviderShared.imageResponse(ADAPTER, "Z.ai Images", response)
|
||||
const decoded = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(ZAIImageResponse))(output.body).pipe(
|
||||
Effect.mapError((cause) => output.invalid("Z.ai Images returned an invalid response", cause)),
|
||||
)
|
||||
if (decoded.data.length === 0) return yield* output.invalid("Z.ai Images returned no images")
|
||||
return new ImageResponse({
|
||||
images: decoded.data.map(
|
||||
(item) =>
|
||||
new GeneratedImage({
|
||||
mediaType: "application/octet-stream",
|
||||
data: item.url,
|
||||
}),
|
||||
),
|
||||
providerMetadata: {
|
||||
zai: {
|
||||
created: decoded.created,
|
||||
id: decoded.id,
|
||||
requestID: decoded.request_id,
|
||||
contentFilter: decoded.content_filter,
|
||||
},
|
||||
},
|
||||
})
|
||||
}),
|
||||
}
|
||||
return ImageModel.make<ZAIImageOptions>({ id: input.id, provider: "zai", route, http: input.http })
|
||||
}
|
||||
// ---------------------------------------------------------------------------
|
||||
// 6. Response decoding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const decodeResponse = Effect.fn("ZAIImages.decodeResponse")(function* (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const output = yield* MediaProtocol.decodeJson(ADAPTER, NAME, ZAIImageResponse)(response)
|
||||
const decoded = output.value
|
||||
if (decoded.data.length === 0) return yield* output.invalid(`${NAME} returned no images`)
|
||||
const filters = decoded.content_filter ?? []
|
||||
return new ImageResponse({
|
||||
// Z.ai returns only URLs and no content type; the media type resolves when the asset is materialized.
|
||||
images: decoded.data.map((item) => Media.url(item.url)),
|
||||
// Z.ai reports applied content filters alongside a successful result; surface them instead of dropping them.
|
||||
notices:
|
||||
filters.length === 0
|
||||
? undefined
|
||||
: filters.map((filter) => ({
|
||||
type: "moderated" as const,
|
||||
message: `${NAME} applied a content filter${filter.role === undefined ? "" : ` for ${filter.role}`}${
|
||||
filter.level === undefined ? "" : ` at level ${filter.level}`
|
||||
}`,
|
||||
providerMetadata: { zai: filter },
|
||||
})),
|
||||
providerMetadata: {
|
||||
zai: {
|
||||
created: decoded.created,
|
||||
id: decoded.id,
|
||||
requestID: decoded.request_id,
|
||||
contentFilter: decoded.content_filter,
|
||||
},
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// 7. Protocol and route
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const protocol = MediaProtocol.inline<Request, ImageResponse>({
|
||||
id: ADAPTER,
|
||||
name: NAME,
|
||||
unsupported: ["images", "mask", "n", "aspectRatio", "seed", "format"],
|
||||
body: { from: fromRequest },
|
||||
response: { decode: decodeResponse },
|
||||
})
|
||||
|
||||
export const model = (input: MediaRoute.ModelInput) =>
|
||||
ImageModel.fromRoute<ZAIImageOptions>(
|
||||
{ id: ADAPTER, provider: PROVIDER, protocol, baseURL: DEFAULT_BASE_URL, path: PATH },
|
||||
input,
|
||||
)
|
||||
|
||||
export const ZAIImages = {
|
||||
protocol,
|
||||
model,
|
||||
} as const
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
import { Auth } from "../route/auth.js"
|
||||
import type { ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { AssemblyAITranscription, DEFAULT_BASE_URL } from "../protocols/assemblyai-transcription.js"
|
||||
|
||||
export type { AssemblyAITranscriptionOptions } from "../protocols/assemblyai-transcription.js"
|
||||
|
||||
export const id = ProviderID.make("assemblyai")
|
||||
const baseURL = DEFAULT_BASE_URL
|
||||
|
||||
export type Config = ProviderAuthOption<"optional"> & {
|
||||
/** `https://api.eu.assemblyai.com` for the EU region. */
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
// The key is the whole `authorization` value, without a scheme.
|
||||
const auth = (options: ProviderAuthOption<"optional">) => {
|
||||
if ("auth" in options && options.auth) return options.auth
|
||||
return Auth.optional("apiKey" in options ? options.apiKey : undefined, "apiKey")
|
||||
.orElse(Auth.config("ASSEMBLYAI_API_KEY"))
|
||||
.pipe(Auth.header("authorization"))
|
||||
}
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const transcription = (modelID: string | ModelID) =>
|
||||
AssemblyAITranscription.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
transcription,
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const transcription = provider.transcription
|
||||
@@ -0,0 +1,35 @@
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { CartesiaSpeech, DEFAULT_BASE_URL } from "../protocols/cartesia-speech.js"
|
||||
|
||||
export type { CartesiaEncoding, CartesiaSpeechOptions } from "../protocols/cartesia-speech.js"
|
||||
|
||||
export const id = ProviderID.make("cartesia")
|
||||
const baseURL = DEFAULT_BASE_URL
|
||||
|
||||
export type Config = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
const auth = (options: ProviderAuthOption<"optional">) => AuthOptions.bearer(options, "CARTESIA_API_KEY")
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const speech = (modelID: string | ModelID) =>
|
||||
CartesiaSpeech.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
speech,
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const speech = provider.speech
|
||||
@@ -44,7 +44,12 @@ export const configure = (input: LanguageModelOptions = {}) => {
|
||||
model: (modelID: string | ModelID) =>
|
||||
configured.model<OpenAIProviderOptionsInput>({
|
||||
id: modelID,
|
||||
compatibility: { maxTokensField: "max_tokens", reasoningField: "reasoning", supportsStore: false, supportsPromptCacheKey: true },
|
||||
compatibility: {
|
||||
maxTokensField: "max_tokens",
|
||||
reasoningField: "reasoning",
|
||||
supportsStore: false,
|
||||
supportsPromptCacheKey: true,
|
||||
},
|
||||
}),
|
||||
configure,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
import { Auth } from "../route/auth.js"
|
||||
import type { ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { DEFAULT_BASE_URL, DeepgramSpeech } from "../protocols/deepgram-speech.js"
|
||||
import { DeepgramTranscription } from "../protocols/deepgram-transcription.js"
|
||||
|
||||
export type { DeepgramEncoding, DeepgramSpeechOptions } from "../protocols/deepgram-speech.js"
|
||||
export type { DeepgramTranscriptionOptions } from "../protocols/deepgram-transcription.js"
|
||||
|
||||
export const id = ProviderID.make("deepgram")
|
||||
const baseURL = DEFAULT_BASE_URL
|
||||
|
||||
export type Config = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
const auth = (options: ProviderAuthOption<"optional">) => {
|
||||
if ("auth" in options && options.auth) return options.auth
|
||||
return Auth.optional("apiKey" in options ? options.apiKey : undefined, "apiKey")
|
||||
.orElse(Auth.config("DEEPGRAM_API_KEY"))
|
||||
.pipe(Auth.scheme("Token"))
|
||||
}
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const media = (modelID: string | ModelID) => ({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
speech: (modelID: string | ModelID) => DeepgramSpeech.model(media(modelID)),
|
||||
transcription: (modelID: string | ModelID) => DeepgramTranscription.model(media(modelID)),
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const speech = provider.speech
|
||||
export const transcription = provider.transcription
|
||||
@@ -47,7 +47,12 @@ export const configure = (input: LanguageModelOptions = {}) => {
|
||||
model: (modelID: string | ModelID) =>
|
||||
configured.model<OpenAIProviderOptionsInput>({
|
||||
id: modelID,
|
||||
compatibility: { maxTokensField: "max_tokens", reasoningField: "reasoning_content", supportsStore: false, supportsPromptCacheKey: true },
|
||||
compatibility: {
|
||||
maxTokensField: "max_tokens",
|
||||
reasoningField: "reasoning_content",
|
||||
supportsStore: false,
|
||||
supportsPromptCacheKey: true,
|
||||
},
|
||||
}),
|
||||
configure,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
import { Auth } from "../route/auth.js"
|
||||
import type { ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { DEFAULT_BASE_URL, ElevenLabsSpeech } from "../protocols/elevenlabs-speech.js"
|
||||
|
||||
export type { ElevenLabsOutputFormat, ElevenLabsSpeechOptions } from "../protocols/elevenlabs-speech.js"
|
||||
|
||||
export const id = ProviderID.make("elevenlabs")
|
||||
const baseURL = DEFAULT_BASE_URL
|
||||
|
||||
export type Config = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
const auth = (options: ProviderAuthOption<"optional">) => {
|
||||
if ("auth" in options && options.auth) return options.auth
|
||||
return Auth.optional("apiKey" in options ? options.apiKey : undefined, "apiKey")
|
||||
.orElse(Auth.config("ELEVENLABS_API_KEY"))
|
||||
.pipe(Auth.header("xi-api-key"))
|
||||
}
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const speech = (modelID: string | ModelID) =>
|
||||
ElevenLabsSpeech.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
speech,
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const speech = provider.speech
|
||||
@@ -0,0 +1,42 @@
|
||||
import { Auth } from "../route/auth.js"
|
||||
import type { ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { DEFAULT_BASE_URL, FalVideo } from "../protocols/fal-video.js"
|
||||
|
||||
export type { FalVideoOptions } from "../protocols/fal-video.js"
|
||||
|
||||
export const id = ProviderID.make("fal")
|
||||
const baseURL = DEFAULT_BASE_URL
|
||||
|
||||
export type Config = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
// fal authenticates with `Authorization: Key <FAL_KEY>` rather than a bearer token.
|
||||
const auth = (options: ProviderAuthOption<"optional">) => {
|
||||
if ("auth" in options && options.auth) return options.auth
|
||||
return Auth.optional("apiKey" in options ? options.apiKey : undefined, "apiKey")
|
||||
.orElse(Auth.config("FAL_KEY"))
|
||||
.pipe(Auth.scheme("Key"))
|
||||
}
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const video = (modelID: string | ModelID) =>
|
||||
FalVideo.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
video,
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const video = provider.video
|
||||
@@ -38,23 +38,13 @@ export type Settings = ProviderPackage.Settings &
|
||||
|
||||
const fromRequest = Effect.fn("GoogleVertex.fromRequest")(function* (request: LLMRequest) {
|
||||
const { serviceTier: _, ...body } = yield* Gemini.protocol.body.from(request)
|
||||
// Vertex's native REST schema rejects `id` on FunctionCall/FunctionResponse parts with HTTP 400,
|
||||
// unlike AI Studio, so history minted there cannot be lowered verbatim.
|
||||
const contents = body.contents.map((content) => ({
|
||||
...content,
|
||||
parts: (content.parts ?? []).map((part) => {
|
||||
if ("functionCall" in part) return { ...part, functionCall: { ...part.functionCall, id: undefined } }
|
||||
if ("functionResponse" in part) return { ...part, functionResponse: { ...part.functionResponse, id: undefined } }
|
||||
return part
|
||||
}),
|
||||
}))
|
||||
const value = request.providerOptions?.labels
|
||||
const labels = ProviderShared.isRecord(value)
|
||||
? Object.fromEntries(
|
||||
Object.entries(value).filter((entry): entry is [string, string] => typeof entry[1] === "string"),
|
||||
)
|
||||
: undefined
|
||||
return { ...body, contents, labels }
|
||||
return { ...body, labels }
|
||||
})
|
||||
|
||||
const protocol = {
|
||||
|
||||
@@ -5,8 +5,14 @@ import type { ProviderPackage } from "../provider-package.js"
|
||||
import { HttpOptions, ProviderID, mergeHttpOptions, type ModelID } from "../schema/index.js"
|
||||
import { Gemini } from "../protocols/gemini.js"
|
||||
import { GoogleImages } from "../protocols/google-images.js"
|
||||
import { GoogleSpeech } from "../protocols/google-speech.js"
|
||||
import { GoogleTranscription } from "../protocols/google-transcription.js"
|
||||
import { GoogleVideo } from "../protocols/google-video.js"
|
||||
|
||||
export type { GoogleImageOptions } from "../protocols/google-images.js"
|
||||
export type { GoogleSpeechOptions } from "../protocols/google-speech.js"
|
||||
export type { GoogleTranscriptionOptions } from "../protocols/google-transcription.js"
|
||||
export type { GoogleVideoOptions } from "../protocols/google-video.js"
|
||||
export type GeminiOptionsInput = Gemini.OptionsInput
|
||||
export type GeminiProviderOptionsInput = Gemini.ProviderOptionsInput
|
||||
|
||||
@@ -40,18 +46,20 @@ const configuredRoute = (input: Config) => {
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const route = configuredRoute(input)
|
||||
const image = (modelID: string | ModelID) =>
|
||||
GoogleImages.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL,
|
||||
headers: input.headers,
|
||||
http: mergeHttpOptions(input.http === undefined ? undefined : HttpOptions.make(input.http)),
|
||||
})
|
||||
const media = (modelID: string | ModelID) => ({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL,
|
||||
headers: input.headers,
|
||||
http: mergeHttpOptions(input.http === undefined ? undefined : HttpOptions.make(input.http)),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
model: (modelID: string | ModelID) => route.model<Gemini.ProviderOptionsInput>({ id: modelID }),
|
||||
image,
|
||||
image: (modelID: string | ModelID) => GoogleImages.model(media(modelID)),
|
||||
video: (modelID: string | ModelID) => GoogleVideo.model(media(modelID)),
|
||||
speech: (modelID: string | ModelID) => GoogleSpeech.model(media(modelID)),
|
||||
transcription: (modelID: string | ModelID) => GoogleTranscription.model(media(modelID)),
|
||||
configure,
|
||||
}
|
||||
}
|
||||
@@ -70,3 +78,6 @@ export const model: ProviderPackage.Definition<Settings, Gemini.ProviderOptionsI
|
||||
}).model(modelID)
|
||||
|
||||
export const image = provider.image
|
||||
export const video = provider.video
|
||||
export const speech = provider.speech
|
||||
export const transcription = provider.transcription
|
||||
|
||||
@@ -3,13 +3,18 @@ export * as Anthropic from "./anthropic.js"
|
||||
export * as AnthropicCompatible from "./anthropic-compatible.js"
|
||||
export * as AmazonBedrock from "./amazon-bedrock.js"
|
||||
export * as AmazonBedrockMantle from "./amazon-bedrock-mantle.js"
|
||||
export * as AssemblyAI from "./assemblyai.js"
|
||||
export * as Azure from "./azure.js"
|
||||
export * as Baseten from "./baseten.js"
|
||||
export * as Cartesia from "./cartesia.js"
|
||||
export * as Cerebras from "./cerebras.js"
|
||||
export * as CloudflareAIGateway from "./cloudflare-ai-gateway.js"
|
||||
export * as CloudflareWorkersAI from "./cloudflare-workers-ai.js"
|
||||
export * as DeepInfra from "./deepinfra.js"
|
||||
export * as Deepgram from "./deepgram.js"
|
||||
export * as DeepSeek from "./deepseek.js"
|
||||
export * as ElevenLabs from "./elevenlabs.js"
|
||||
export * as Fal from "./fal.js"
|
||||
export * as Fireworks from "./fireworks.js"
|
||||
export * as Google from "./google.js"
|
||||
export * as GoogleVertex from "./google-vertex.js"
|
||||
@@ -24,8 +29,12 @@ export * as Moonshot from "./moonshot.js"
|
||||
export * as OpenAI from "./openai.js"
|
||||
export * as OpenAICompatible from "./openai-compatible.js"
|
||||
export * as OpenAICompatibleResponses from "./openai-compatible-responses.js"
|
||||
export * as OpenCodeZen from "./opencode-zen.js"
|
||||
export * as OpenRouter from "./openrouter.js"
|
||||
export * as Runway from "./runway.js"
|
||||
export * as TogetherAI from "./togetherai.js"
|
||||
export * as TypeSafeAI from "./typesafe-ai.js"
|
||||
export * as VercelAIGateway from "./vercel-ai-gateway.js"
|
||||
export * as XAI from "./xai.js"
|
||||
export * as ZAI from "./zai.js"
|
||||
export * as ZAICodingPlan from "./zai-coding-plan.js"
|
||||
|
||||
@@ -6,9 +6,13 @@ import * as OpenAIChat from "../protocols/openai-chat.js"
|
||||
import * as OpenAIResponses from "../protocols/openai-responses.js"
|
||||
import { withOpenAIOptions, type OpenAIProviderOptionsInput } from "./openai-options.js"
|
||||
import { OpenAIImages, type OpenAIImageString } from "../protocols/openai-images.js"
|
||||
import { OpenAISpeech } from "../protocols/openai-speech.js"
|
||||
import { OpenAITranscription } from "../protocols/openai-transcription.js"
|
||||
|
||||
export type { OpenAIOptionsInput, OpenAIResponseIncludable } from "./openai-options.js"
|
||||
export type { OpenAIImageOptions } from "../protocols/openai-images.js"
|
||||
export type { OpenAISpeechOptions } from "../protocols/openai-speech.js"
|
||||
export type { OpenAITranscriptionOptions } from "../protocols/openai-transcription.js"
|
||||
|
||||
export const id = ProviderID.make("openai")
|
||||
|
||||
@@ -95,17 +99,19 @@ export const configure = (input: Config = {}) => {
|
||||
id,
|
||||
compatibility: { supportsPromptCacheKey: true },
|
||||
})
|
||||
const image = (modelID: string | ModelID) =>
|
||||
OpenAIImages.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL,
|
||||
headers: input.headers,
|
||||
http: mergeHttpOptions(
|
||||
input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
input.queryParams === undefined ? undefined : new HttpOptions({ query: input.queryParams }),
|
||||
),
|
||||
})
|
||||
const media = (modelID: string | ModelID) => ({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL,
|
||||
headers: input.headers,
|
||||
http: mergeHttpOptions(
|
||||
input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
input.queryParams === undefined ? undefined : new HttpOptions({ query: input.queryParams }),
|
||||
),
|
||||
})
|
||||
const image = (modelID: string | ModelID) => OpenAIImages.model(media(modelID))
|
||||
const speech = (modelID: string | ModelID) => OpenAISpeech.model(media(modelID))
|
||||
const transcription = (modelID: string | ModelID) => OpenAITranscription.model(media(modelID))
|
||||
|
||||
return {
|
||||
id,
|
||||
@@ -113,6 +119,8 @@ export const configure = (input: Config = {}) => {
|
||||
responses,
|
||||
chat,
|
||||
image,
|
||||
speech,
|
||||
transcription,
|
||||
configure,
|
||||
}
|
||||
}
|
||||
@@ -159,3 +167,5 @@ export const chatModel: ProviderPackage.Definition<Settings, OpenAIProviderOptio
|
||||
export const responses = provider.responses
|
||||
export const chat = provider.chat
|
||||
export const image = provider.image
|
||||
export const speech = provider.speech
|
||||
export const transcription = provider.transcription
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { SystemOne } from "../experimental/system-one.js"
|
||||
|
||||
export const id = ProviderID.make("opencode")
|
||||
const baseURL = "https://opencode.ai/zen/v1"
|
||||
|
||||
export type Options = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
export const configure = (input: Options = {}) => {
|
||||
const evaluation = (modelID: string | ModelID) =>
|
||||
SystemOne.model({
|
||||
id: modelID,
|
||||
provider: id,
|
||||
providerMetadataKey: "opencode",
|
||||
auth: AuthOptions.bearer(input, "OPENCODE_API_KEY"),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return { id, experimental: { evaluation }, configure }
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const experimental = provider.experimental
|
||||
|
||||
export * as OpenCodeZen from "./opencode-zen.js"
|
||||
@@ -3,8 +3,9 @@ import { Route, type RouteDefaultsInput } from "../route/client.js"
|
||||
import { Endpoint } from "../route/endpoint.js"
|
||||
import { Protocol } from "../route/protocol.js"
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { ProviderID, type CacheHint, type ModelID } from "../schema/index.js"
|
||||
import { HttpOptions, ProviderID, type CacheHint, type ModelID } from "../schema/index.js"
|
||||
import type { ProviderPackage } from "../provider-package.js"
|
||||
import { SystemOne } from "../experimental/system-one.js"
|
||||
import { OpenAIChat } from "../protocols/openai-chat.js"
|
||||
import { newBreakpoints, ttlBucket } from "../protocols/utils/cache.js"
|
||||
import { isRecord } from "../protocols/shared.js"
|
||||
@@ -71,6 +72,14 @@ export interface OpenRouterOptions {
|
||||
|
||||
export type OpenRouterProviderOptionsInput = OpenRouterOptions
|
||||
|
||||
export interface OpenRouterEvaluationOptions {
|
||||
readonly [key: string]: unknown
|
||||
readonly provider?: OpenRouterProviderRouting
|
||||
readonly session_id?: string
|
||||
readonly trace?: Readonly<Record<string, unknown>>
|
||||
readonly user?: string
|
||||
}
|
||||
|
||||
export type LanguageModelOptions = Omit<RouteDefaultsInput, "providerOptions"> &
|
||||
ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
@@ -181,15 +190,27 @@ const configuredRoute = (input: LanguageModelOptions) => {
|
||||
|
||||
export const configure = (input: LanguageModelOptions = {}) => {
|
||||
const route = configuredRoute(input)
|
||||
const evaluation = (modelID: string | ModelID) =>
|
||||
SystemOne.model<OpenRouterEvaluationOptions>({
|
||||
id: modelID,
|
||||
provider: id,
|
||||
providerMetadataKey: "openrouter",
|
||||
auth: AuthOptions.bearer(input, "OPENROUTER_API_KEY"),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
model: (modelID: string | ModelID) =>
|
||||
route.model<OpenRouterProviderOptionsInput>({ id: modelID, compatibility: { supportsPromptCacheKey: true } }),
|
||||
experimental: { evaluation },
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const experimental = provider.experimental
|
||||
export const model: ProviderPackage.Definition<Settings, OpenRouterProviderOptionsInput>["model"] = (
|
||||
modelID,
|
||||
{ apiKey, baseURL, body, headers, ...providerOptions },
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { DEFAULT_BASE_URL, RunwayVideo } from "../protocols/runway-video.js"
|
||||
|
||||
export type { RunwayVideoOptions } from "../protocols/runway-video.js"
|
||||
|
||||
export const id = ProviderID.make("runway")
|
||||
const baseURL = DEFAULT_BASE_URL
|
||||
|
||||
export type Config = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
const auth = (options: ProviderAuthOption<"optional">) => AuthOptions.bearer(options, "RUNWAYML_API_SECRET")
|
||||
|
||||
export const configure = (input: Config = {}) => {
|
||||
const video = (modelID: string | ModelID) =>
|
||||
RunwayVideo.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
video,
|
||||
configure,
|
||||
}
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const video = provider.video
|
||||
@@ -0,0 +1,31 @@
|
||||
import { HttpOptions, ProviderID, type ModelID } from "../schema/index.js"
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
|
||||
import { SystemOne } from "../experimental/system-one.js"
|
||||
|
||||
export const id = ProviderID.make("typesafe-ai")
|
||||
const baseURL = "https://api.typesafe.ai/v1"
|
||||
|
||||
export type Options = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
export const configure = (input: Options = {}) => {
|
||||
const evaluation = (modelID: string | ModelID) =>
|
||||
SystemOne.model({
|
||||
id: modelID,
|
||||
provider: id,
|
||||
providerMetadataKey: "typesafe",
|
||||
auth: AuthOptions.bearer(input, "TYPESAFE_API_KEY"),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return { id, experimental: { evaluation }, configure }
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const experimental = provider.experimental
|
||||
|
||||
export * as TypeSafeAI from "./typesafe-ai.js"
|
||||
@@ -0,0 +1,148 @@
|
||||
import { Effect, Schema } from "effect"
|
||||
import { Headers, HttpClientRequest } from "effect/unstable/http"
|
||||
import {
|
||||
EvaluationAnswer,
|
||||
EvaluationInput,
|
||||
EvaluationModel,
|
||||
EvaluationQuestion,
|
||||
EvaluationResponse,
|
||||
EvaluationRounding,
|
||||
} from "../experimental/evaluation.js"
|
||||
import { Auth } from "../route/auth.js"
|
||||
import { AuthOptions, type ProviderAuthOption } from "../route/auth-options.js"
|
||||
import {
|
||||
AIError,
|
||||
HttpContext,
|
||||
HttpOptions,
|
||||
InvalidProviderOutputError,
|
||||
InvalidRequestError,
|
||||
ModelID,
|
||||
ProviderID,
|
||||
ProviderMetadata,
|
||||
Usage,
|
||||
} from "../schema/index.js"
|
||||
|
||||
export const id = ProviderID.make("vercel-ai-gateway")
|
||||
const baseURL = "https://ai-gateway.vercel.sh/v1"
|
||||
|
||||
export interface EvaluationOptions {
|
||||
readonly [key: string]: unknown
|
||||
readonly gateway?: Readonly<{
|
||||
readonly [key: string]: unknown
|
||||
readonly zeroDataRetention?: boolean
|
||||
readonly only?: ReadonlyArray<string>
|
||||
}>
|
||||
}
|
||||
|
||||
export type Options = ProviderAuthOption<"optional"> & {
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
const Request = Schema.StructWithRest(
|
||||
Schema.Struct({
|
||||
model: Schema.String,
|
||||
state: EvaluationInput,
|
||||
questions: Schema.Record(Schema.String, EvaluationQuestion),
|
||||
providerOptions: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
}),
|
||||
[Schema.Record(Schema.String, Schema.Any)],
|
||||
)
|
||||
const Response = Schema.Struct({
|
||||
model: Schema.optional(Schema.String),
|
||||
answers: Schema.Record(Schema.String, EvaluationAnswer),
|
||||
usage: Schema.optional(
|
||||
Schema.Struct({
|
||||
inputTokens: Schema.optional(Schema.Number),
|
||||
outputTokens: Schema.optional(Schema.Number),
|
||||
}),
|
||||
),
|
||||
rounding: Schema.optional(EvaluationRounding),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
})
|
||||
|
||||
export const configure = (input: Options = {}) => {
|
||||
const evaluation = (modelID: string | ModelID) =>
|
||||
EvaluationModel.make<EvaluationOptions>({
|
||||
id: modelID,
|
||||
provider: id,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
route: {
|
||||
id: "vercel-evaluation",
|
||||
evaluate: (req, send) =>
|
||||
Effect.gen(function* () {
|
||||
const url = new URL(`${(input.baseURL ?? baseURL).replace(/\/$/, "")}/evaluate`)
|
||||
Object.entries(req.http?.query ?? {}).forEach(([key, value]) => url.searchParams.set(key, value))
|
||||
const body = yield* Schema.encodeUnknownEffect(Schema.fromJsonString(Request))({
|
||||
...req.http?.body,
|
||||
model: req.model.id,
|
||||
state: req.state,
|
||||
questions: req.questions,
|
||||
providerOptions: req.options,
|
||||
}).pipe(
|
||||
Effect.mapError(
|
||||
(cause) => new AIError({ reason: new InvalidRequestError({ message: cause.message, cause }) }),
|
||||
),
|
||||
)
|
||||
const headers = yield* Auth.toEffect(
|
||||
AuthOptions.bearer(input, ["AI_GATEWAY_API_KEY", "VERCEL_OIDC_TOKEN"]),
|
||||
)({
|
||||
request: req,
|
||||
method: "POST",
|
||||
url: url.toString(),
|
||||
body,
|
||||
headers: Headers.fromInput({ ...input.headers, ...req.http?.headers }),
|
||||
})
|
||||
const res = yield* send(
|
||||
HttpClientRequest.post(url).pipe(
|
||||
HttpClientRequest.setHeaders(headers),
|
||||
HttpClientRequest.bodyText(body, "application/json"),
|
||||
),
|
||||
)
|
||||
const http = new HttpContext({ url: res.request.url, status: res.status, headers: res.headers })
|
||||
const fail = (message: string, cause: unknown, body?: string) =>
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({
|
||||
route: "vercel-evaluation",
|
||||
message,
|
||||
body,
|
||||
http,
|
||||
cause,
|
||||
}),
|
||||
})
|
||||
const text = yield* res.text.pipe(
|
||||
Effect.mapError((cause) => fail("Failed to read the Vercel AI Gateway evaluation response", cause)),
|
||||
)
|
||||
const data = yield* Schema.decodeUnknownEffect(Schema.fromJsonString(Response))(text).pipe(
|
||||
Effect.mapError((cause) =>
|
||||
fail("Vercel AI Gateway returned an invalid evaluation response", cause, text),
|
||||
),
|
||||
)
|
||||
return new EvaluationResponse({
|
||||
model: ModelID.make(data.model ?? req.model.id),
|
||||
answers: data.answers,
|
||||
usage: data.usage
|
||||
? new Usage({
|
||||
inputTokens: data.usage.inputTokens,
|
||||
outputTokens: data.usage.outputTokens,
|
||||
totalTokens:
|
||||
data.usage.inputTokens === undefined && data.usage.outputTokens === undefined
|
||||
? undefined
|
||||
: (data.usage.inputTokens ?? 0) + (data.usage.outputTokens ?? 0),
|
||||
providerMetadata: { gateway: data.usage },
|
||||
})
|
||||
: undefined,
|
||||
rounding: data.rounding,
|
||||
providerMetadata: data.providerMetadata,
|
||||
})
|
||||
}),
|
||||
},
|
||||
})
|
||||
return { id, experimental: { evaluation }, configure }
|
||||
}
|
||||
|
||||
export const provider = configure()
|
||||
export const experimental = provider.experimental
|
||||
|
||||
export * as VercelAIGateway from "./vercel-ai-gateway.js"
|
||||
@@ -6,6 +6,7 @@ import { OpenAIChat } from "../protocols/openai-chat.js"
|
||||
import { OpenResponsesChannel } from "../protocols/open-responses-channel.js"
|
||||
import { XAIResponses } from "../protocols/xai-responses.js"
|
||||
import { XAIImages } from "../protocols/xai-images.js"
|
||||
import { XAIVideo } from "../protocols/xai-video.js"
|
||||
import type { OpenAIOptionsInput } from "./openai-options.js"
|
||||
import type { ProviderPackage } from "../provider-package.js"
|
||||
|
||||
@@ -27,6 +28,7 @@ export type Settings = ProviderPackage.Settings &
|
||||
}
|
||||
|
||||
export type { XAIImageOptions } from "../protocols/xai-images.js"
|
||||
export type { XAIVideoOptions } from "../protocols/xai-video.js"
|
||||
|
||||
const RESPONSES_WEBSOCKET_ROTATE_AFTER_MS = 24 * 60 * 1000
|
||||
|
||||
@@ -87,20 +89,20 @@ export const configure = (input: LanguageModelOptions = {}) => {
|
||||
const chatRoute = configuredChatRoute(input)
|
||||
const responses = (modelID: string | ModelID) => responsesRoute.model<XAIProviderOptionsInput>({ id: modelID })
|
||||
const chat = (modelID: string | ModelID) => chatRoute.model<XAIProviderOptionsInput>({ id: modelID })
|
||||
const image = (modelID: string | ModelID) =>
|
||||
XAIImages.model({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
const media = (modelID: string | ModelID) => ({
|
||||
id: modelID,
|
||||
auth: auth(input),
|
||||
baseURL: input.baseURL ?? baseURL,
|
||||
headers: input.headers,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
return {
|
||||
id,
|
||||
model: responses,
|
||||
responses,
|
||||
chat,
|
||||
image,
|
||||
image: (modelID: string | ModelID) => XAIImages.model(media(modelID)),
|
||||
video: (modelID: string | ModelID) => XAIVideo.model(media(modelID)),
|
||||
configure,
|
||||
}
|
||||
}
|
||||
@@ -121,3 +123,4 @@ export const model: ProviderPackage.Definition<
|
||||
export const responses = provider.responses
|
||||
export const chat = provider.chat
|
||||
export const image = provider.image
|
||||
export const video = provider.video
|
||||
|
||||
@@ -16,7 +16,7 @@ type Secret = string | Redacted.Redacted | Config.Config<string | Redacted.Redac
|
||||
|
||||
export interface AuthInput {
|
||||
readonly request: { readonly http?: HttpOptions }
|
||||
readonly method: "POST" | "GET"
|
||||
readonly method: "POST" | "GET" | "PUT" | "DELETE"
|
||||
readonly url: string
|
||||
readonly body: string
|
||||
readonly headers: Headers.Headers
|
||||
@@ -134,6 +134,16 @@ export function bearerHeader(name: string, source?: Secret | Credential) {
|
||||
return render(source)
|
||||
}
|
||||
|
||||
/** `Authorization: <scheme> <secret>` for providers whose scheme is not `Bearer`, such as fal's `Key`. */
|
||||
export function scheme(name: string): (source: Secret | Credential) => Definition
|
||||
export function scheme(name: string, source: Secret | Credential): Definition
|
||||
export function scheme(name: string, source?: Secret | Credential) {
|
||||
const render = (input: Secret | Credential) =>
|
||||
fromCredential(credentialInput(input), (secret) => ({ authorization: `${name} ${secret}` }))
|
||||
if (source === undefined) return render
|
||||
return render(source)
|
||||
}
|
||||
|
||||
const toAIError = (error: AuthError): AIError => {
|
||||
if (error instanceof MissingCredentialError || error instanceof Config.ConfigError) {
|
||||
return new AIError({
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
import type { LLMRequest } from "../schema/index.js"
|
||||
import * as ProviderShared from "../protocols/shared.js"
|
||||
|
||||
export interface EndpointInput<Body> {
|
||||
readonly request: LLMRequest
|
||||
export interface EndpointInput<Body, Request = LLMRequest> {
|
||||
readonly request: Request
|
||||
readonly body: Body
|
||||
}
|
||||
|
||||
export type EndpointPart<Body> = string | ((input: EndpointInput<Body>) => string)
|
||||
export type EndpointPart<Body, Request = LLMRequest> = string | ((input: EndpointInput<Body, Request>) => string)
|
||||
|
||||
/**
|
||||
* Declarative URL construction for one route.
|
||||
@@ -17,26 +17,29 @@ export type EndpointPart<Body> = string | ((input: EndpointInput<Body>) => strin
|
||||
*
|
||||
* `path` may be a string or a function of `EndpointInput`, for routes whose
|
||||
* URL embeds the model id, region, or another body field (e.g. Bedrock,
|
||||
* Gemini).
|
||||
* Gemini). Media routes reuse the same shape with their own request type.
|
||||
*/
|
||||
export interface Definition<Body> {
|
||||
export interface Definition<Body, Request = LLMRequest> {
|
||||
readonly baseURL?: string
|
||||
readonly path: EndpointPart<Body>
|
||||
readonly path: EndpointPart<Body, Request>
|
||||
readonly query?: Record<string, string>
|
||||
}
|
||||
|
||||
export type EndpointPatch<Body> = Partial<Definition<Body>>
|
||||
export type EndpointPatch<Body, Request = LLMRequest> = Partial<Definition<Body, Request>>
|
||||
|
||||
/** Construct an `Endpoint` from a path string or path function. */
|
||||
export const path = <Body>(
|
||||
value: EndpointPart<Body>,
|
||||
options: Omit<Definition<Body>, "path"> = {},
|
||||
): Definition<Body> => ({
|
||||
export const path = <Body, Request = LLMRequest>(
|
||||
value: EndpointPart<Body, Request>,
|
||||
options: Omit<Definition<Body, Request>, "path"> = {},
|
||||
): Definition<Body, Request> => ({
|
||||
...options,
|
||||
path: value,
|
||||
})
|
||||
|
||||
export const merge = <Body>(base: Definition<Body>, patch: EndpointPatch<Body>): Definition<Body> => ({
|
||||
export const merge = <Body, Request = LLMRequest>(
|
||||
base: Definition<Body, Request>,
|
||||
patch: EndpointPatch<Body, Request>,
|
||||
): Definition<Body, Request> => ({
|
||||
...base,
|
||||
...patch,
|
||||
baseURL: patch.baseURL ?? base.baseURL,
|
||||
@@ -44,10 +47,13 @@ export const merge = <Body>(base: Definition<Body>, patch: EndpointPatch<Body>):
|
||||
query: patch.query === undefined ? base.query : { ...base.query, ...patch.query },
|
||||
})
|
||||
|
||||
const renderPart = <Body>(part: EndpointPart<Body>, input: EndpointInput<Body>) =>
|
||||
const renderPart = <Body, Request>(part: EndpointPart<Body, Request>, input: EndpointInput<Body, Request>) =>
|
||||
typeof part === "function" ? part(input) : part
|
||||
|
||||
export const render = <Body>(endpoint: Definition<Body>, input: EndpointInput<Body>) => {
|
||||
export const render = <Body, Request = LLMRequest>(
|
||||
endpoint: Definition<Body, Request>,
|
||||
input: EndpointInput<Body, Request>,
|
||||
) => {
|
||||
const url = new URL(`${ProviderShared.trimBaseUrl(endpoint.baseURL ?? "")}${renderPart(endpoint.path, input)}`)
|
||||
for (const [key, value] of Object.entries(endpoint.query ?? {})) url.searchParams.set(key, value)
|
||||
return url
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
import { Context, type Effect } from "effect"
|
||||
import type { HttpClientRequest, HttpClientResponse } from "effect/unstable/http"
|
||||
import type { AIError } from "../schema/errors.js"
|
||||
|
||||
// The service tag lives in its own leaf module so `Media.Asset` (imported by the schema layer) can require the
|
||||
// executor without pulling the full executor implementation, and therefore the schema barrel, into a cycle.
|
||||
export interface Interface {
|
||||
readonly execute: (
|
||||
request: HttpClientRequest.HttpClientRequest,
|
||||
middleware?: HttpMiddleware,
|
||||
) => Effect.Effect<HttpClientResponse.HttpClientResponse, AIError>
|
||||
}
|
||||
|
||||
export type HttpHandler = (
|
||||
request: HttpClientRequest.HttpClientRequest,
|
||||
) => Effect.Effect<HttpClientResponse.HttpClientResponse, Error>
|
||||
export type HttpMiddleware = (
|
||||
request: HttpClientRequest.HttpClientRequest,
|
||||
handler: HttpHandler,
|
||||
) => Effect.Effect<HttpClientResponse.HttpClientResponse, Error>
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/AI/RequestExecutor") {}
|
||||
@@ -1,4 +1,4 @@
|
||||
import { Cause, Context, Effect, Layer, Option, Schema, Stream } from "effect"
|
||||
import { Cause, Effect, Layer, Option, Schema, Stream } from "effect"
|
||||
import {
|
||||
FetchHttpClient,
|
||||
Headers,
|
||||
@@ -9,23 +9,10 @@ import {
|
||||
} from "effect/unstable/http"
|
||||
import { HttpContext, HttpRateLimitDetails, AIError, TransportError } from "../schema/index.js"
|
||||
import { classifyProviderFailure } from "../provider-error.js"
|
||||
import { Service, type HttpMiddleware, type Interface } from "./executor-service.js"
|
||||
|
||||
export interface Interface {
|
||||
readonly execute: (
|
||||
request: HttpClientRequest.HttpClientRequest,
|
||||
middleware?: HttpMiddleware,
|
||||
) => Effect.Effect<HttpClientResponse.HttpClientResponse, AIError>
|
||||
}
|
||||
|
||||
export type HttpHandler = (
|
||||
request: HttpClientRequest.HttpClientRequest,
|
||||
) => Effect.Effect<HttpClientResponse.HttpClientResponse, Error>
|
||||
export type HttpMiddleware = (
|
||||
request: HttpClientRequest.HttpClientRequest,
|
||||
handler: HttpHandler,
|
||||
) => Effect.Effect<HttpClientResponse.HttpClientResponse, Error>
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/AI/RequestExecutor") {}
|
||||
export { Service } from "./executor-service.js"
|
||||
export type { HttpHandler, HttpMiddleware, Interface } from "./executor-service.js"
|
||||
|
||||
const headerDetails = (headers: Headers.Headers) =>
|
||||
Object.fromEntries(Object.entries(headers).map(([name, value]) => [name, String(value)]))
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import type { Stream } from "effect"
|
||||
import { Stream } from "effect"
|
||||
import * as ProviderShared from "../protocols/shared.js"
|
||||
import type { AIError } from "../schema/index.js"
|
||||
|
||||
@@ -12,6 +12,8 @@ import type { AIError } from "../schema/index.js"
|
||||
* `[DONE]`; protocols that use it as a terminal select `sseWithDone`.
|
||||
* - AWS event stream — length-prefixed binary frames with CRC checksums.
|
||||
* Each emitted frame is one parsed binary event record.
|
||||
* - Media streams — newline-delimited JSON (`lines`) or the whole body as one
|
||||
* frame (`document`); chunked binary bodies need no framing.
|
||||
*
|
||||
* The frame type is opaque to this layer; the protocol's event schema decodes
|
||||
* each frame before its state machine handles it.
|
||||
@@ -38,4 +40,19 @@ export const sseEvents = (events: ReadonlySet<string>): Definition<string> => ({
|
||||
frame: (bytes) => ProviderShared.sseFraming(bytes, events),
|
||||
})
|
||||
|
||||
export const lines: Definition<string> = {
|
||||
id: "lines",
|
||||
frame: (bytes) =>
|
||||
bytes.pipe(
|
||||
Stream.decodeText(),
|
||||
Stream.splitLines,
|
||||
Stream.filter((line) => line.trim().length > 0),
|
||||
),
|
||||
}
|
||||
|
||||
export const document: Definition<string> = {
|
||||
id: "document",
|
||||
frame: (bytes) => Stream.fromEffect(Stream.mkString(bytes.pipe(Stream.decodeText()))),
|
||||
}
|
||||
|
||||
export * as Framing from "./framing.js"
|
||||
|
||||
@@ -20,6 +20,8 @@ export * from "./executor.js"
|
||||
export { Auth } from "./auth.js"
|
||||
export { AuthOptions } from "./auth-options.js"
|
||||
export { Endpoint } from "./endpoint.js"
|
||||
export { MediaRoute } from "./media.js"
|
||||
export { MediaProtocol } from "./media-protocol.js"
|
||||
export { Framing } from "./framing.js"
|
||||
export { Protocol } from "./protocol.js"
|
||||
export { HttpTransport, WebSocketTransport } from "./transport/index.js"
|
||||
|
||||
@@ -0,0 +1,295 @@
|
||||
import { Clock, Duration, Effect, Schema, type Stream } from "effect"
|
||||
import { HttpClientResponse } from "effect/unstable/http"
|
||||
import type { Snapshot, Status } from "../generation.js"
|
||||
import { Media } from "../media.js"
|
||||
import type { AuthInput } from "./auth.js"
|
||||
import {
|
||||
AIError,
|
||||
ContentPolicyError,
|
||||
HttpContext,
|
||||
InvalidProviderOutputError,
|
||||
InvalidRequestError,
|
||||
ProviderInternalError,
|
||||
} from "../schema/index.js"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Bodies
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Array values become repeated parameters (`keyterm=a&keyterm=b`). */
|
||||
export type Query = Readonly<Record<string, string | ReadonlyArray<string>>>
|
||||
|
||||
/** `query` is appended to the endpoint URL before the route and caller `http.query` overlays. */
|
||||
export type Body =
|
||||
| { readonly type: "json"; readonly value: Record<string, unknown>; readonly query?: Query }
|
||||
| { readonly type: "multipart"; readonly value: FormData }
|
||||
| {
|
||||
readonly type: "binary"
|
||||
readonly value: Uint8Array
|
||||
readonly contentType: string
|
||||
readonly query?: Query
|
||||
}
|
||||
|
||||
export const json = (value: Record<string, unknown>, query?: Query): Body => ({
|
||||
type: "json",
|
||||
value,
|
||||
query,
|
||||
})
|
||||
export const multipart = (value: FormData): Body => ({ type: "multipart", value })
|
||||
export const binary = (value: Uint8Array, contentType: string, query?: Query): Body => ({
|
||||
type: "binary",
|
||||
value,
|
||||
contentType,
|
||||
query,
|
||||
})
|
||||
|
||||
export type Send = (path: string, body: Body) => Effect.Effect<HttpClientResponse.HttpClientResponse, AIError>
|
||||
|
||||
/** Runs after unsupported-field rejection and before `body.from`, for providers that need an upload first. */
|
||||
export type Prepare<Request> = (request: Request, send: Send) => Effect.Effect<Request, AIError>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Protocol kinds
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export interface DecodeContext<Request> {
|
||||
readonly request: Request
|
||||
readonly body: Body
|
||||
}
|
||||
|
||||
/** One request, one response. JSON or multipart in; JSON or raw bytes out. */
|
||||
export interface Inline<Request, Response> {
|
||||
readonly kind: "inline"
|
||||
readonly id: string
|
||||
readonly name: string
|
||||
/** Common request fields this protocol cannot lower; the route rejects them before `body.from` runs. */
|
||||
readonly unsupported?: ReadonlyArray<keyof Request & string>
|
||||
readonly body: { readonly from: (request: Request) => Effect.Effect<Body, AIError> }
|
||||
readonly response: {
|
||||
readonly decode: (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: DecodeContext<Request>,
|
||||
) => Effect.Effect<Response, AIError>
|
||||
}
|
||||
}
|
||||
|
||||
export const inline = <Request, Response>(
|
||||
input: Omit<Inline<Request, Response>, "kind">,
|
||||
): Inline<Request, Response> => ({
|
||||
kind: "inline",
|
||||
...input,
|
||||
})
|
||||
|
||||
/** What `start` learned from the submission response: the route-owned handle plus the first observation. */
|
||||
export interface Started<Token> {
|
||||
readonly token: Token
|
||||
readonly snapshot: Snapshot
|
||||
}
|
||||
|
||||
/**
|
||||
* A follow-up call's inputs: the decoded token and the auth headers the route sent, so a protocol can attach them
|
||||
* to output URLs that require the same credentials to download (Veo).
|
||||
*/
|
||||
export interface PollContext<Token> {
|
||||
readonly token: Token
|
||||
readonly auth: Record<string, string>
|
||||
}
|
||||
|
||||
/**
|
||||
* Submit, then poll. `start` posts the body to the route endpoint; `status`, `result`, and `cancel` are follow-up
|
||||
* calls addressed by the token. Paths are relative to the route base URL unless the provider hands back absolute
|
||||
* URLs (fal `status_url`), in which case they are used verbatim. `result` is always its own GET: providers that
|
||||
* return the output inside the status body (Veo, xAI, Runway) point `result.path` at the status path and decode the
|
||||
* same document, so `Generation.await` and `Video.resume(...).await()` behave identically everywhere.
|
||||
*/
|
||||
export interface Queued<Request, Response, Token> {
|
||||
readonly kind: "queued"
|
||||
readonly id: string
|
||||
readonly name: string
|
||||
/** Common request fields this protocol cannot lower; the route rejects them before `start.body.from` runs. */
|
||||
readonly unsupported?: ReadonlyArray<keyof Request & string>
|
||||
/** Serializable handle. `Generation.token` carries the encoded form so it can be persisted and resumed elsewhere. */
|
||||
readonly token: Schema.Codec<Token, unknown>
|
||||
readonly start: {
|
||||
readonly prepare?: Prepare<Request>
|
||||
readonly body: { readonly from: (request: Request) => Effect.Effect<Body, AIError> }
|
||||
readonly decode: (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: DecodeContext<Request>,
|
||||
) => Effect.Effect<Started<Token>, AIError>
|
||||
}
|
||||
readonly status: {
|
||||
readonly path: (token: Token) => string
|
||||
readonly decode: (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: PollContext<Token>,
|
||||
) => Effect.Effect<Snapshot, AIError>
|
||||
}
|
||||
readonly result: {
|
||||
readonly path: (token: Token) => string
|
||||
readonly decode: (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: PollContext<Token>,
|
||||
) => Effect.Effect<Response, AIError>
|
||||
}
|
||||
readonly cancel?: {
|
||||
readonly method: AuthInput["method"]
|
||||
readonly path: (token: Token) => string
|
||||
}
|
||||
}
|
||||
|
||||
export const queued = <Request, Response, Token>(
|
||||
input: Omit<Queued<Request, Response, Token>, "kind">,
|
||||
): Queued<Request, Response, Token> => ({
|
||||
kind: "queued",
|
||||
...input,
|
||||
})
|
||||
|
||||
export type Mode = "generate" | "stream"
|
||||
|
||||
export type Addressed<Request> = Request & { readonly mode: Mode }
|
||||
|
||||
export interface ResponseContext<Request> extends DecodeContext<Addressed<Request>> {
|
||||
readonly http: HttpContext
|
||||
}
|
||||
|
||||
/**
|
||||
* One request whose body is parsed incrementally, like LLM protocols: `frames` → `step`* → `finish`. `generate` and
|
||||
* `stream` share this state machine; `request.mode` lets a protocol pick a different body, path, or framing.
|
||||
*/
|
||||
export interface Streamed<Request, Event, Frame, State> {
|
||||
readonly kind: "stream"
|
||||
readonly id: string
|
||||
readonly name: string
|
||||
/** Common request fields this protocol cannot lower; the route rejects them before `body.from` runs. */
|
||||
readonly unsupported?: ReadonlyArray<keyof Request & string>
|
||||
readonly body: { readonly from: (request: Addressed<Request>) => Effect.Effect<Body, AIError> }
|
||||
readonly frames: (
|
||||
bytes: Stream.Stream<Uint8Array, AIError>,
|
||||
context: DecodeContext<Addressed<Request>>,
|
||||
) => Stream.Stream<Frame, AIError>
|
||||
readonly initial: () => State
|
||||
readonly step: (state: State, frame: Frame) => Effect.Effect<readonly [State, ReadonlyArray<Event>], AIError>
|
||||
/** Emit exactly one terminal event, or fail when the provider stopped before completing. */
|
||||
readonly finish: (state: State, context: ResponseContext<Request>) => Effect.Effect<ReadonlyArray<Event>, AIError>
|
||||
}
|
||||
|
||||
export const stream = <Request, Event, Frame, State>(
|
||||
input: Omit<Streamed<Request, Event, Frame, State>, "kind">,
|
||||
): Streamed<Request, Event, Frame, State> => ({
|
||||
kind: "stream",
|
||||
...input,
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Response helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const context = (response: HttpClientResponse.HttpClientResponse) =>
|
||||
new HttpContext({ url: response.request.url, status: response.status, headers: response.headers })
|
||||
|
||||
/**
|
||||
* Read a text body while retaining the original payload and HTTP context on every downstream error. `invalid` is a
|
||||
* malformed provider document; `ended` is a generation that reached a terminal status without output (`failed` is
|
||||
* provider-side, `cancelled`/`expired` mean the result will never exist); `contentPolicy` is a moderated result.
|
||||
*/
|
||||
export const text = Effect.fn("MediaProtocol.text")(function* (
|
||||
route: string,
|
||||
name: string,
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
) {
|
||||
const http = context(response)
|
||||
const body = yield* response.text.pipe(
|
||||
Effect.mapError(
|
||||
(cause) =>
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({
|
||||
route,
|
||||
message: `Failed to read the ${name} response`,
|
||||
http,
|
||||
cause,
|
||||
}),
|
||||
}),
|
||||
),
|
||||
)
|
||||
return {
|
||||
body,
|
||||
http,
|
||||
invalid: (message: string, cause?: unknown) =>
|
||||
new AIError({ reason: new InvalidProviderOutputError({ route, message, body, http, cause }) }),
|
||||
ended: (status: Exclude<Status, "queued" | "running" | "completed">, message: string) =>
|
||||
new AIError({
|
||||
reason:
|
||||
status === "failed"
|
||||
? new ProviderInternalError({ message, body, http })
|
||||
: new InvalidRequestError({ message, body, http }),
|
||||
}),
|
||||
contentPolicy: (message: string) => new AIError({ reason: new ContentPolicyError({ message, body, http }) }),
|
||||
}
|
||||
})
|
||||
|
||||
export type Output = Effect.Success<ReturnType<typeof text>>
|
||||
|
||||
/** Read and Schema-decode a JSON body. Decode failures keep the raw body as `reason.body`. */
|
||||
export const decodeJson = <A>(route: string, name: string, schema: Schema.Codec<A, unknown>) => {
|
||||
const decode = Schema.decodeUnknownEffect(Schema.fromJsonString(schema))
|
||||
return Effect.fn("MediaProtocol.decodeJson")(function* (response: HttpClientResponse.HttpClientResponse) {
|
||||
const output = yield* text(route, name, response)
|
||||
const value = yield* decode(output.body).pipe(
|
||||
Effect.mapError((cause) => output.invalid(`${name} returned an invalid response`, cause)),
|
||||
)
|
||||
return { ...output, value }
|
||||
})
|
||||
}
|
||||
|
||||
/** Decode a submission response into the token and first snapshot. */
|
||||
export const decodeStarted = <A, Token>(
|
||||
route: string,
|
||||
name: string,
|
||||
schema: Schema.Codec<A, unknown>,
|
||||
started: (value: A) => Started<Token>,
|
||||
) => {
|
||||
const decode = decodeJson(route, name, schema)
|
||||
return (response: HttpClientResponse.HttpClientResponse) =>
|
||||
decode(response).pipe(Effect.map((output) => started(output.value)))
|
||||
}
|
||||
|
||||
/** Map a provider status string through the protocol's table; unknown values are an invalid provider document. */
|
||||
export const status = <Table extends Record<string, Status>>(
|
||||
table: Table,
|
||||
raw: string,
|
||||
output: Output,
|
||||
): Effect.Effect<Status, AIError> => {
|
||||
const normalized: Status | undefined = table[raw]
|
||||
if (normalized === undefined) return Effect.fail(output.invalid(`Unknown generation status "${raw}"`))
|
||||
return Effect.succeed(normalized)
|
||||
}
|
||||
|
||||
export const frameError = (route: string, message: string, body?: string, cause?: unknown) =>
|
||||
new AIError({ reason: new InvalidProviderOutputError({ route, message, body, cause }) })
|
||||
|
||||
export const incomplete = (route: string) =>
|
||||
new AIError({
|
||||
reason: new InvalidProviderOutputError({
|
||||
route,
|
||||
message: "The provider response ended unexpectedly.",
|
||||
classification: "incomplete-stream",
|
||||
}),
|
||||
})
|
||||
|
||||
/** Schema-decode one JSON stream frame. Decode failures keep the frame as `reason.body`. */
|
||||
export const decodeFrame = <A>(route: string, name: string, schema: Schema.Codec<A, unknown>) => {
|
||||
const decode = Schema.decodeUnknownEffect(Schema.fromJsonString(schema))
|
||||
return (frame: string) =>
|
||||
decode(frame).pipe(
|
||||
Effect.mapError((cause) => frameError(route, `${name} sent an invalid stream event`, frame, cause)),
|
||||
)
|
||||
}
|
||||
|
||||
/** A `url` asset whose provider-declared retention window starts now. */
|
||||
export const expiringUrl = (url: string, retention: Duration.Duration, options?: Parameters<typeof Media.url>[1]) =>
|
||||
Clock.currentTimeMillis.pipe(
|
||||
Effect.map((now) => Media.url(url, { ...options, expiresAt: now + Duration.toMillis(retention) })),
|
||||
)
|
||||
|
||||
export * as MediaProtocol from "./media-protocol.js"
|
||||
@@ -0,0 +1,385 @@
|
||||
import { Effect, Schema, Stream } from "effect"
|
||||
import { Headers, HttpClientRequest, type HttpClientResponse } from "effect/unstable/http"
|
||||
import { Auth, type AuthInput } from "./auth.js"
|
||||
import { Endpoint } from "./endpoint.js"
|
||||
import type { Interface } from "./executor-service.js"
|
||||
import { RequestExecutor } from "./executor.js"
|
||||
import { MediaProtocol } from "./media-protocol.js"
|
||||
import { Generation, type Route as GenerationRoute } from "../generation.js"
|
||||
import { ProviderShared } from "../protocols/shared.js"
|
||||
import {
|
||||
AIError,
|
||||
AIErrorReason,
|
||||
HttpOptions,
|
||||
InvalidRequestError,
|
||||
ProviderID,
|
||||
mergeHttpOptions,
|
||||
} from "../schema/index.js"
|
||||
import { sanitizeSurrogates } from "../utils/sanitize.js"
|
||||
|
||||
export type Execute = Interface["execute"]
|
||||
|
||||
/** The minimum a media request must carry for the route to build a transport request. */
|
||||
export interface MediaRequest {
|
||||
readonly model: { readonly id: string; readonly provider: ProviderID; readonly http?: HttpOptions }
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
|
||||
/** Deployment inputs every media model factory accepts; provider facades fill these from `configure(...)`. */
|
||||
export interface ModelInput {
|
||||
readonly id: string
|
||||
readonly auth: Auth.Definition
|
||||
readonly baseURL?: string
|
||||
readonly headers?: Record<string, string>
|
||||
readonly http?: HttpOptions
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Routes
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** One request, one response. */
|
||||
export interface Route<Request extends MediaRequest, Response> {
|
||||
readonly kind: "inline"
|
||||
readonly id: string
|
||||
readonly provider: ProviderID
|
||||
readonly protocol: string
|
||||
readonly generate: (request: Request, execute: Execute) => Effect.Effect<Response, AIError>
|
||||
}
|
||||
|
||||
/** Submit, then poll through the returned `Generation`. */
|
||||
export interface QueuedRoute<Request extends MediaRequest, Response> {
|
||||
readonly kind: "queued"
|
||||
readonly id: string
|
||||
readonly provider: ProviderID
|
||||
readonly protocol: string
|
||||
readonly start: (request: Request, execute: Execute) => Effect.Effect<Generation<Response>, AIError>
|
||||
/** Rebuild a handle from a persisted `Generation.token`; fails typed when the token is not this route's. */
|
||||
readonly resume: (
|
||||
model: MediaRequest["model"],
|
||||
token: unknown,
|
||||
execute: Execute,
|
||||
) => Effect.Effect<Generation<Response>, AIError>
|
||||
}
|
||||
|
||||
/** One request whose response parses into events; `generate` runs the same stream and collects it. */
|
||||
export interface StreamRoute<Request extends MediaRequest, Event, Response> {
|
||||
readonly kind: "stream"
|
||||
readonly id: string
|
||||
readonly provider: ProviderID
|
||||
readonly protocol: string
|
||||
readonly stream: (request: Request, execute: Execute) => Stream.Stream<Event, AIError>
|
||||
readonly generate: (request: Request, execute: Execute) => Effect.Effect<Response, AIError>
|
||||
}
|
||||
|
||||
export interface Composition<Request extends MediaRequest> {
|
||||
readonly id: string
|
||||
readonly provider: string | ProviderID
|
||||
readonly endpoint: Endpoint.Definition<MediaProtocol.Body, Request>
|
||||
readonly auth: Auth.Definition
|
||||
/** Deployment headers applied before transport authentication. */
|
||||
readonly headers?: Record<string, string>
|
||||
}
|
||||
|
||||
export interface InlineInput<Request extends MediaRequest, Response> extends Composition<Request> {
|
||||
readonly protocol: MediaProtocol.Inline<Request, Response>
|
||||
}
|
||||
|
||||
export interface QueuedInput<Request extends MediaRequest, Response, Token> extends Composition<Request> {
|
||||
readonly protocol: MediaProtocol.Queued<Request, Response, Token>
|
||||
}
|
||||
|
||||
export interface StreamInput<Request extends MediaRequest, Event, Response, Frame, State>
|
||||
extends Composition<MediaProtocol.Addressed<Request>> {
|
||||
readonly protocol: MediaProtocol.Streamed<Request, Event, Frame, State>
|
||||
readonly collect: (events: ReadonlyArray<Event>) => Effect.Effect<Response, AIError>
|
||||
}
|
||||
|
||||
/**
|
||||
* Compose an inline media protocol with an endpoint and auth into a runnable route. The route owns the transport
|
||||
* plumbing every media protocol would otherwise duplicate: option merging, surrogate sanitizing, unsupported-field
|
||||
* rejection, URL and query rendering, auth headers, JSON, multipart, or binary encoding, and handing responses back
|
||||
* to the protocol.
|
||||
*/
|
||||
export const inline = <Request extends MediaRequest, Response>(
|
||||
input: InlineInput<Request, Response>,
|
||||
): Route<Request, Response> => {
|
||||
const transport = makeTransport(input)
|
||||
return {
|
||||
kind: "inline",
|
||||
id: input.id,
|
||||
provider: transport.provider,
|
||||
protocol: input.protocol.id,
|
||||
generate: Effect.fn(`MediaRoute.generate`)(function* (request: Request, execute: Execute) {
|
||||
const submitted = yield* transport.submit(
|
||||
request,
|
||||
{ unsupported: input.protocol.unsupported, from: input.protocol.body.from },
|
||||
execute,
|
||||
)
|
||||
return yield* input.protocol.response.decode(submitted.response, submitted.context)
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compose a queued media protocol the same way, adding `start`/`resume` handles whose polls reuse the route's auth,
|
||||
* deployment headers, and (for `start`) the request's `http` overlay. The token is decoded once at the boundary and
|
||||
* closed over by the resulting `Generation.Route`.
|
||||
*/
|
||||
export const queued = <Request extends MediaRequest, Response, Token>(
|
||||
input: QueuedInput<Request, Response, Token>,
|
||||
): QueuedRoute<Request, Response> => {
|
||||
const transport = makeTransport(input)
|
||||
const protocol = input.protocol
|
||||
const decodeToken = Schema.decodeUnknownEffect(protocol.token)
|
||||
// A protocol producing a token its own codec rejects is a programmer defect, not a provider error.
|
||||
const encodeToken = Schema.encodeSync(protocol.token)
|
||||
|
||||
const generationRoute = (token: Token, http: HttpOptions | undefined, execute: Execute) => {
|
||||
const poll = <A>(operation: {
|
||||
readonly path: (token: Token) => string
|
||||
readonly decode: (
|
||||
response: HttpClientResponse.HttpClientResponse,
|
||||
context: MediaProtocol.PollContext<Token>,
|
||||
) => Effect.Effect<A, AIError>
|
||||
}) =>
|
||||
transport
|
||||
.call("GET", operation.path(token), http, execute)
|
||||
.pipe(Effect.flatMap((sent) => operation.decode(sent.response, { token, auth: sent.auth })))
|
||||
const cancel = protocol.cancel
|
||||
const route: GenerationRoute<Response> = {
|
||||
status: poll(protocol.status),
|
||||
result: poll(protocol.result),
|
||||
cancel:
|
||||
cancel === undefined
|
||||
? undefined
|
||||
: transport.call(cancel.method, cancel.path(token), http, execute).pipe(Effect.asVoid),
|
||||
}
|
||||
return route
|
||||
}
|
||||
|
||||
const start = Effect.fn("MediaRoute.start")(function* (request: Request, execute: Execute) {
|
||||
const submitted = yield* transport.submit(
|
||||
request,
|
||||
{ unsupported: protocol.unsupported, prepare: protocol.start.prepare, from: protocol.start.body.from },
|
||||
execute,
|
||||
)
|
||||
const started = yield* protocol.start.decode(submitted.response, submitted.context)
|
||||
const route = generationRoute(started.token, submitted.context.request.http, execute)
|
||||
return new Generation(route, encodeToken(started.token), started.snapshot)
|
||||
})
|
||||
|
||||
const resume = Effect.fn("MediaRoute.resume")(function* (
|
||||
model: MediaRequest["model"],
|
||||
raw: unknown,
|
||||
execute: Execute,
|
||||
) {
|
||||
const token = yield* decodeToken(raw).pipe(
|
||||
Effect.mapError(
|
||||
(cause) =>
|
||||
new AIError({
|
||||
reason: new InvalidRequestError({
|
||||
message: `${input.id} cannot resume a generation from this token`,
|
||||
cause,
|
||||
}),
|
||||
}),
|
||||
),
|
||||
)
|
||||
const route = generationRoute(token, transport.http(model), execute)
|
||||
return new Generation(route, encodeToken(token), yield* route.status)
|
||||
})
|
||||
|
||||
return { kind: "queued", id: input.id, provider: transport.provider, protocol: protocol.id, start, resume }
|
||||
}
|
||||
|
||||
/** Compose a streaming media protocol; `generate` runs the same stream in `generate` mode and folds it with `collect`. */
|
||||
export const stream = <Request extends MediaRequest, Event, Response, Frame, State>(
|
||||
input: StreamInput<Request, Event, Response, Frame, State>,
|
||||
): StreamRoute<Request, Event, Response> => {
|
||||
const transport = makeTransport(input)
|
||||
const protocol = input.protocol
|
||||
const events = (request: Request, execute: Execute, mode: MediaProtocol.Mode) =>
|
||||
Stream.unwrap(
|
||||
Effect.gen(function* () {
|
||||
const submitted = yield* transport.submit(
|
||||
{ ...request, mode },
|
||||
{ unsupported: protocol.unsupported, from: protocol.body.from },
|
||||
execute,
|
||||
)
|
||||
const http = RequestExecutor.responseHttp(submitted.response)
|
||||
return Stream.suspend(() => {
|
||||
// Parser state is local to one response, exactly like `Route.make`'s LLM stream loop.
|
||||
let state = protocol.initial()
|
||||
return protocol.frames(RequestExecutor.responseStream(submitted.response), submitted.context).pipe(
|
||||
Stream.mapEffect((frame) =>
|
||||
protocol.step(state, frame).pipe(
|
||||
Effect.map(([next, output]) => {
|
||||
state = next
|
||||
return output
|
||||
}),
|
||||
),
|
||||
),
|
||||
Stream.flattenIterable,
|
||||
Stream.concat(
|
||||
Stream.suspend(() => Stream.fromIterableEffect(protocol.finish(state, { ...submitted.context, http }))),
|
||||
),
|
||||
Stream.mapError((error) =>
|
||||
error.reason.http !== undefined
|
||||
? error
|
||||
: new AIError({
|
||||
reason: AIErrorReason.make({
|
||||
...error.reason,
|
||||
message: error.reason.message,
|
||||
cause: error.reason.cause,
|
||||
http,
|
||||
}),
|
||||
}),
|
||||
),
|
||||
)
|
||||
})
|
||||
}),
|
||||
)
|
||||
return {
|
||||
kind: "stream",
|
||||
id: input.id,
|
||||
provider: transport.provider,
|
||||
protocol: protocol.id,
|
||||
stream: (request, execute) => events(request, execute, "stream"),
|
||||
generate: (request, execute) =>
|
||||
events(request, execute, "generate").pipe(Stream.runCollect, Effect.flatMap(input.collect)),
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Transport plumbing shared by every kind
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const makeTransport = <Request extends MediaRequest>(input: Composition<Request>) => {
|
||||
const provider = ProviderID.make(input.provider)
|
||||
const routeHttp = input.headers === undefined ? undefined : new HttpOptions({ headers: input.headers })
|
||||
const authorize = Auth.toEffect(input.auth)
|
||||
const baseURL = (path: string) => new URL(`${ProviderShared.trimBaseUrl(input.endpoint.baseURL ?? "")}${path}`)
|
||||
/** `auth` is only what `Auth` added, never deployment headers. */
|
||||
const send = Effect.fn("MediaRoute.send")(function* (
|
||||
call: {
|
||||
readonly method: AuthInput["method"]
|
||||
readonly url: URL
|
||||
readonly headers: Headers.Headers
|
||||
readonly request: AuthInput["request"]
|
||||
readonly body?: MediaProtocol.Body
|
||||
},
|
||||
execute: Execute,
|
||||
) {
|
||||
const encoded = encode(call.body, call.headers)
|
||||
const url = call.url.toString()
|
||||
const headers = yield* authorize({
|
||||
request: call.request,
|
||||
method: call.method,
|
||||
url,
|
||||
body: encoded.text,
|
||||
headers: encoded.headers,
|
||||
})
|
||||
const response = yield* execute(
|
||||
encoded.apply(HttpClientRequest.make(call.method)(url).pipe(HttpClientRequest.setHeaders(headers))),
|
||||
)
|
||||
return { response, auth: Object.fromEntries(Object.entries(headers).filter(([key]) => !(key in call.headers))) }
|
||||
})
|
||||
return {
|
||||
provider,
|
||||
/** Route and model overlays; `start` additionally merges the request's own `http`. */
|
||||
http: (model: MediaRequest["model"]) => mergeHttpOptions(routeHttp, model.http),
|
||||
/** POST the protocol body to the route endpoint. */
|
||||
submit: Effect.fn("MediaRoute.submit")(function* (
|
||||
request: Request,
|
||||
protocol: {
|
||||
readonly unsupported?: ReadonlyArray<keyof Request & string>
|
||||
readonly prepare?: MediaProtocol.Prepare<Request>
|
||||
readonly from: (request: Request) => Effect.Effect<MediaProtocol.Body, AIError>
|
||||
},
|
||||
execute: Execute,
|
||||
) {
|
||||
yield* rejectUnsupported(input.id, provider, request, protocol.unsupported)
|
||||
const http = mergeHttpOptions(routeHttp, request.model.http, request.http)
|
||||
const headers = Headers.fromInput(http?.headers)
|
||||
const prepared =
|
||||
protocol.prepare === undefined
|
||||
? request
|
||||
: yield* protocol.prepare(request, (path, body) =>
|
||||
send({ method: "POST", url: baseURL(path), headers, request, body }, execute).pipe(
|
||||
Effect.map((sent) => sent.response),
|
||||
),
|
||||
)
|
||||
// Sanitize after merging so model-level overlays are covered; the model value is restored, not sanitized.
|
||||
const resolved: Request = { ...sanitizeSurrogates({ ...prepared, http }), model: request.model }
|
||||
const body = yield* protocol.from(resolved)
|
||||
const url = withQuery(
|
||||
withQuery(
|
||||
Endpoint.render(input.endpoint, { request: resolved, body }),
|
||||
body.type === "multipart" ? undefined : body.query,
|
||||
),
|
||||
http?.query,
|
||||
)
|
||||
const sent = yield* send({ method: "POST", url, headers, request: resolved, body }, execute)
|
||||
return { response: sent.response, context: { request: resolved, body } }
|
||||
}),
|
||||
/** Bodiless follow-up call (status, result, cancel) with the same auth and headers as `submit`. */
|
||||
call: (method: AuthInput["method"], path: string, http: HttpOptions | undefined, execute: Execute) => {
|
||||
// Provider-issued absolute URLs (fal `status_url`) are used as-is; everything else resolves against the base.
|
||||
const url = withQuery(/^https?:\/\//.test(path) ? new URL(path) : baseURL(path), http?.query)
|
||||
for (const [key, value] of Object.entries(input.endpoint.query ?? {})) url.searchParams.set(key, value)
|
||||
return send({ method, url, headers: Headers.fromInput(http?.headers), request: { http } }, execute)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const withQuery = (url: URL, query: MediaProtocol.Query | undefined) => {
|
||||
for (const [key, value] of Object.entries(query ?? {})) {
|
||||
url.searchParams.delete(key)
|
||||
for (const item of typeof value === "string" ? [value] : value) url.searchParams.append(key, item)
|
||||
}
|
||||
return url
|
||||
}
|
||||
|
||||
const encode = (body: MediaProtocol.Body | undefined, headers: Headers.Headers) => {
|
||||
if (body === undefined) return { text: "", headers, apply: (request: HttpClientRequest.HttpClientRequest) => request }
|
||||
if (body.type === "json") {
|
||||
const text = ProviderShared.encodeJson(body.value)
|
||||
return { text, headers, apply: HttpClientRequest.bodyText(text, "application/json") }
|
||||
}
|
||||
if (body.type === "binary")
|
||||
return {
|
||||
text: `[${body.contentType}]`,
|
||||
headers,
|
||||
apply: HttpClientRequest.bodyUint8Array(body.value, body.contentType),
|
||||
}
|
||||
return {
|
||||
text: "[multipart/form-data]",
|
||||
// The HTTP client sets the multipart boundary; a caller-supplied content-type would corrupt it.
|
||||
headers: Headers.remove(headers, "content-type"),
|
||||
apply: HttpClientRequest.bodyFormData(body.value),
|
||||
}
|
||||
}
|
||||
|
||||
/** Common fields are never silently dropped: a present field the protocol declared unsupported fails typed. */
|
||||
const rejectUnsupported = <Request extends object>(
|
||||
route: string,
|
||||
provider: ProviderID,
|
||||
request: Request,
|
||||
unsupported: ReadonlyArray<keyof Request & string> | undefined,
|
||||
): Effect.Effect<void, AIError> => {
|
||||
const present = (unsupported ?? []).filter((field) => {
|
||||
const value = request[field]
|
||||
return Array.isArray(value) ? value.length > 0 : value !== undefined
|
||||
})
|
||||
if (present.length === 0) return Effect.void
|
||||
return Effect.fail(
|
||||
ProviderShared.unsupportedOperation({
|
||||
operation: `media.${present[0]}`,
|
||||
provider,
|
||||
route,
|
||||
message: `${provider}/${route} does not support ${present.join(", ")}`,
|
||||
}),
|
||||
)
|
||||
}
|
||||
|
||||
export * as MediaRoute from "./media.js"
|
||||
@@ -215,7 +215,9 @@ export const fromWebSocket = (
|
||||
): Effect.Effect<WebSocketConnection, AIError> =>
|
||||
Effect.gen(function* () {
|
||||
yield* waitOpen(ws, input)
|
||||
const messages = yield* Queue.bounded<string | Uint8Array, AIError | Cause.Done<void>>(128)
|
||||
// The socket pushes frames synchronously and cannot be paused, so the hand-off to the consumer
|
||||
// fiber must absorb whole read buffers. Bun delivers over a thousand small frames in one tick.
|
||||
const messages = yield* Queue.unbounded<string | Uint8Array, AIError | Cause.Done<void>>()
|
||||
|
||||
const oversized = (message: string | Uint8Array) =>
|
||||
typeof message === "string" ? new Blob([message]).size > MAX_FRAME_BYTES : message.byteLength > MAX_FRAME_BYTES
|
||||
@@ -238,19 +240,7 @@ export const fromWebSocket = (
|
||||
}
|
||||
const offer = (message: string | Uint8Array) => {
|
||||
if (rejectOversized(message)) return
|
||||
if (Queue.offerUnsafe(messages, message)) return
|
||||
Queue.failCauseUnsafe(
|
||||
messages,
|
||||
Cause.fail(
|
||||
transportError("WebSocket inbound queue overflow", {
|
||||
body: typeof message === "string" ? message : new TextDecoder().decode(message),
|
||||
url: input.url,
|
||||
operation: "read",
|
||||
code: "queue-overflow",
|
||||
phase: "receive",
|
||||
}),
|
||||
),
|
||||
)
|
||||
Queue.offerUnsafe(messages, message)
|
||||
}
|
||||
|
||||
const onMessage = (event: MessageEvent) => {
|
||||
|
||||
@@ -133,6 +133,12 @@ export class UnknownProviderError extends Schema.TaggedError<UnknownProviderErro
|
||||
ReasonFields,
|
||||
) {}
|
||||
|
||||
/** A caller-supplied deadline elapsed, such as `Generation.await` polling past its `Poll.timeout`. */
|
||||
export class TimeoutError extends Schema.TaggedError<TimeoutError>("AI.Error.Timeout")("Timeout", {
|
||||
...ReasonFields,
|
||||
timeoutMs: Schema.optional(Schema.Number),
|
||||
}) {}
|
||||
|
||||
export const AIErrorReason = Schema.Union([
|
||||
InvalidRequestError,
|
||||
UnsupportedOperationError,
|
||||
@@ -145,6 +151,7 @@ export const AIErrorReason = Schema.Union([
|
||||
TransportError,
|
||||
InvalidProviderOutputError,
|
||||
UnknownProviderError,
|
||||
TimeoutError,
|
||||
]).pipe(Schema.toTaggedUnion("_tag"))
|
||||
export type AIErrorReason = Schema.Schema.Type<typeof AIErrorReason>
|
||||
|
||||
|
||||
@@ -4,18 +4,18 @@ import { ContentBlockID, ToolCallID } from "./ids.js"
|
||||
import {
|
||||
Message,
|
||||
CompactionPart,
|
||||
ProviderMetadata,
|
||||
ToolCallPart,
|
||||
ToolOutput,
|
||||
ToolResultPart,
|
||||
ToolResultValue,
|
||||
type ContentPart,
|
||||
} from "./messages.js"
|
||||
import { ProviderMetadata } from "./options.js"
|
||||
import { ProviderFailureClassification } from "./errors.js"
|
||||
import { Media } from "../media.js"
|
||||
|
||||
export const FinishReason = LLM.FinishReason
|
||||
export type FinishReason = Schema.Schema.Type<typeof FinishReason>
|
||||
export { ProviderMetadata } from "./messages.js"
|
||||
|
||||
/**
|
||||
* Token usage reported by an LLM provider.
|
||||
@@ -91,6 +91,27 @@ export class Usage extends Schema.Class<Usage>("AI.Usage")({
|
||||
|
||||
export type UsageInput = Usage | ConstructorParameters<typeof Usage>[0]
|
||||
|
||||
/**
|
||||
* Usage reported by media routes. Providers bill images, video, speech, and transcription in different units, so
|
||||
* each response carries the unit it was actually metered in instead of forcing everything into tokens.
|
||||
*/
|
||||
export const MediaUsage = Schema.Union([
|
||||
Schema.Struct({
|
||||
type: Schema.Literal("tokens"),
|
||||
input: Schema.optional(Schema.Number),
|
||||
output: Schema.optional(Schema.Number),
|
||||
total: Schema.optional(Schema.Number),
|
||||
details: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
}),
|
||||
Schema.Struct({ type: Schema.Literal("seconds"), seconds: Schema.Number }),
|
||||
Schema.Struct({ type: Schema.Literal("characters"), characters: Schema.Number }),
|
||||
Schema.Struct({ type: Schema.Literal("credits"), credits: Schema.Number }),
|
||||
Schema.Struct({ type: Schema.Literal("compute"), seconds: Schema.Number }),
|
||||
])
|
||||
.pipe(Schema.toTaggedUnion("type"))
|
||||
.annotate({ identifier: "AI.MediaUsage" })
|
||||
export type MediaUsage = Schema.Schema.Type<typeof MediaUsage>
|
||||
|
||||
/** A replacement context window, not an assistant message to append to prior history. */
|
||||
export class CompactionResponse extends Schema.Class<CompactionResponse>("LLM.CompactionResponse")({
|
||||
replacement: Schema.Array(Message),
|
||||
@@ -263,6 +284,14 @@ export const Finish = Schema.Struct({
|
||||
}).annotate({ identifier: "LLM.Event.Finish" })
|
||||
export type Finish = Schema.Schema.Type<typeof Finish>
|
||||
|
||||
/** A generated media asset (image, audio, …) emitted by the model as first-class output rather than a tool result. */
|
||||
export const MediaEvent = Schema.Struct({
|
||||
type: Schema.tag("media"),
|
||||
media: Media.AssetSchema,
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}).annotate({ identifier: "LLM.Event.Media" })
|
||||
export type MediaEvent = Schema.Schema.Type<typeof MediaEvent>
|
||||
|
||||
export const ProviderErrorEvent = Schema.Struct({
|
||||
type: Schema.tag("provider-error"),
|
||||
message: Schema.String,
|
||||
@@ -287,6 +316,7 @@ const llmEventTagged = Schema.Union([
|
||||
ToolCall,
|
||||
ToolResult,
|
||||
ToolError,
|
||||
MediaEvent,
|
||||
StepFinish,
|
||||
Finish,
|
||||
ProviderErrorEvent,
|
||||
@@ -332,6 +362,7 @@ export const LLMEvent = Object.assign(llmEventTagged, {
|
||||
output: input.output === undefined ? undefined : ToolOutput.make(input.output.structured, input.output.content),
|
||||
}),
|
||||
toolError: (input: WithID<ToolError, ToolCallID>) => ToolError.make({ ...input, id: toolCallID(input.id) }),
|
||||
media: MediaEvent.make,
|
||||
stepFinish: (input: WithUsage<StepFinish>) =>
|
||||
StepFinish.make({
|
||||
...input,
|
||||
@@ -359,6 +390,7 @@ export const LLMEvent = Object.assign(llmEventTagged, {
|
||||
toolCall: llmEventTagged.guards["tool-call"],
|
||||
toolResult: llmEventTagged.guards["tool-result"],
|
||||
toolError: llmEventTagged.guards["tool-error"],
|
||||
media: llmEventTagged.guards.media,
|
||||
stepFinish: llmEventTagged.guards["step-finish"],
|
||||
finish: llmEventTagged.guards.finish,
|
||||
providerError: llmEventTagged.guards["provider-error"],
|
||||
@@ -634,6 +666,13 @@ const reduceResponseState = (state: ResponseState, event: LLMEvent): ResponseSta
|
||||
return reduceToolCall(next, event)
|
||||
case "tool-result":
|
||||
return appendContent(next, toolResultContent(event))
|
||||
case "media":
|
||||
return appendContent(
|
||||
next,
|
||||
event.providerMetadata === undefined
|
||||
? { type: "media", media: event.media }
|
||||
: { type: "media", media: event.media, providerMetadata: event.providerMetadata },
|
||||
)
|
||||
default:
|
||||
return next
|
||||
}
|
||||
|
||||
@@ -8,19 +8,16 @@ import {
|
||||
JsonSchema,
|
||||
LanguageModelSchema,
|
||||
type LanguageModel,
|
||||
ProviderMetadata,
|
||||
ProviderOptions,
|
||||
ReasoningEffort,
|
||||
} from "./options.js"
|
||||
import { ProviderID } from "./ids.js"
|
||||
import { Media } from "../media.js"
|
||||
|
||||
export const MessageRole = Schema.Literals(["system", "user", "assistant", "tool"])
|
||||
export type MessageRole = Schema.Schema.Type<typeof MessageRole>
|
||||
|
||||
export const ProviderMetadata = Schema.Record(Schema.String, Schema.Record(Schema.String, Schema.Unknown)).annotate({
|
||||
identifier: "LLM.ProviderMetadata",
|
||||
})
|
||||
export type ProviderMetadata = Schema.Schema.Type<typeof ProviderMetadata>
|
||||
|
||||
const systemPartSchema = Schema.Struct({
|
||||
type: Schema.Literal("text"),
|
||||
text: Schema.String,
|
||||
@@ -50,8 +47,7 @@ export type TextPart = Schema.Schema.Type<typeof TextPart>
|
||||
|
||||
export const MediaPart = Schema.Struct({
|
||||
type: Schema.Literal("media"),
|
||||
mediaType: Schema.String,
|
||||
data: Schema.Union([Schema.String, Schema.Uint8Array]),
|
||||
media: Media.AssetSchema,
|
||||
filename: Schema.optional(Schema.String),
|
||||
cache: Schema.optional(CacheHint),
|
||||
metadata: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
@@ -255,6 +251,12 @@ export namespace Message {
|
||||
|
||||
export const text = (value: string): ContentPart => ({ type: "text", text: value })
|
||||
|
||||
export const media = (asset: Media.Asset, options?: Omit<MediaPart, "type" | "media">): MediaPart => ({
|
||||
type: "media",
|
||||
media: asset,
|
||||
...options,
|
||||
})
|
||||
|
||||
export const content = (input: ContentInput) =>
|
||||
typeof input === "string" ? [text(input)] : Array.isArray(input) ? [...input] : [input]
|
||||
|
||||
|
||||
@@ -39,6 +39,11 @@ const mergeStringRecords = (
|
||||
export const ProviderOptions = Schema.Record(Schema.String, Schema.Unknown)
|
||||
export type ProviderOptions = Schema.Schema.Type<typeof ProviderOptions>
|
||||
|
||||
export const ProviderMetadata = Schema.Record(Schema.String, Schema.Record(Schema.String, Schema.Unknown)).annotate({
|
||||
identifier: "LLM.ProviderMetadata",
|
||||
})
|
||||
export type ProviderMetadata = Schema.Schema.Type<typeof ProviderMetadata>
|
||||
|
||||
export const mergeProviderOptions = (
|
||||
...items: ReadonlyArray<ProviderOptions | undefined>
|
||||
): ProviderOptions | undefined => mergeJsonRecords(...items)
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
import { Context, Effect, Layer, Stream } from "effect"
|
||||
import { RequestExecutor } from "./route/executor.js"
|
||||
import type { AIError } from "./schema/index.js"
|
||||
import type { SpeechEvent, SpeechOptions, SpeechRequestFor, SpeechResponse } from "./speech.js"
|
||||
|
||||
export interface Interface {
|
||||
readonly generate: <Options extends SpeechOptions>(
|
||||
request: SpeechRequestFor<Options>,
|
||||
) => Effect.Effect<SpeechResponse, AIError>
|
||||
readonly stream: <Options extends SpeechOptions>(
|
||||
request: SpeechRequestFor<Options>,
|
||||
) => Stream.Stream<SpeechEvent, AIError>
|
||||
}
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/SpeechClient") {}
|
||||
|
||||
export const generate = <Options extends SpeechOptions>(
|
||||
request: SpeechRequestFor<Options>,
|
||||
): Effect.Effect<SpeechResponse, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.generate(request)
|
||||
})
|
||||
|
||||
export const stream = <Options extends SpeechOptions>(
|
||||
request: SpeechRequestFor<Options>,
|
||||
): Stream.Stream<SpeechEvent, AIError, Service> =>
|
||||
Stream.unwrap(
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return client.stream(request)
|
||||
}),
|
||||
)
|
||||
|
||||
export const layer: Layer.Layer<Service, never, RequestExecutor.Service> = Layer.effect(
|
||||
Service,
|
||||
Effect.gen(function* () {
|
||||
const executor = yield* RequestExecutor.Service
|
||||
return Service.of({
|
||||
generate: (request) => request.model.route.generate(request, executor.execute),
|
||||
stream: (request) => request.model.route.stream(request, executor.execute),
|
||||
})
|
||||
}),
|
||||
)
|
||||
|
||||
export const SpeechClient = {
|
||||
Service,
|
||||
layer,
|
||||
generate,
|
||||
stream,
|
||||
} as const
|
||||
@@ -0,0 +1,217 @@
|
||||
import { Effect, Schema, Stream } from "effect"
|
||||
import { Media } from "./media.js"
|
||||
import { MediaModel, composeRoute, tryRequest } from "./media-model.js"
|
||||
import { MediaRoute } from "./route/media.js"
|
||||
import type { MediaProtocol } from "./route/media-protocol.js"
|
||||
import { AIError, HttpOptions, MediaUsage, ProviderMetadata } from "./schema/index.js"
|
||||
import { SpeechClient, Service } from "./speech-client.js"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Model
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type SpeechOptions = Record<string, unknown>
|
||||
|
||||
export type SpeechRoute<Options extends SpeechOptions = SpeechOptions> = MediaRoute.StreamRoute<
|
||||
SpeechRequestFor<Options>,
|
||||
SpeechEvent,
|
||||
SpeechResponse
|
||||
>
|
||||
|
||||
export class SpeechModel<Options extends SpeechOptions = SpeechOptions> extends MediaModel<
|
||||
SpeechRoute<Options>,
|
||||
Options
|
||||
> {
|
||||
declare protected readonly _SpeechModel: void
|
||||
|
||||
static make<Options extends SpeechOptions = SpeechOptions>(input: MediaModel.Input<SpeechRoute<Options>>) {
|
||||
return new SpeechModel<Options>(input)
|
||||
}
|
||||
|
||||
/** Compose a streaming speech protocol with its canonical path into a model for one deployment. */
|
||||
static fromRoute<Options extends SpeechOptions = SpeechOptions, Frame = unknown, State = unknown>(
|
||||
route: SpeechModel.RouteInput<Options, Frame, State>,
|
||||
input: MediaRoute.ModelInput,
|
||||
) {
|
||||
return new SpeechModel<Options>({
|
||||
id: input.id,
|
||||
provider: route.provider,
|
||||
http: input.http,
|
||||
route: composeRoute(
|
||||
(composition) => MediaRoute.stream({ ...composition, collect: collectResponse }),
|
||||
route,
|
||||
input,
|
||||
),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export namespace SpeechModel {
|
||||
export type RouteInput<
|
||||
Options extends SpeechOptions = SpeechOptions,
|
||||
Frame = unknown,
|
||||
State = unknown,
|
||||
> = MediaModel.RouteInput<
|
||||
MediaProtocol.Addressed<SpeechRequestFor<Options>>,
|
||||
MediaProtocol.Streamed<SpeechRequestFor<Options>, SpeechEvent, Frame, State>
|
||||
>
|
||||
}
|
||||
|
||||
export const SpeechModelSchema = Schema.declare((value): value is SpeechModel => value instanceof SpeechModel, {
|
||||
expected: "Speech.Model",
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Provider-native: a name on OpenAI and Gemini, a voice id on ElevenLabs and Cartesia; `{ id }` is an OpenAI custom voice. */
|
||||
export const SpeechVoice = Schema.Union([Schema.String, Schema.Struct({ id: Schema.String })]).annotate({
|
||||
identifier: "Speech.Voice",
|
||||
})
|
||||
export type SpeechVoice = Schema.Schema.Type<typeof SpeechVoice>
|
||||
|
||||
export type SpeechFormat = "mp3" | "wav" | "pcm" | "opus" | "aac" | "flac" | (string & {})
|
||||
|
||||
/** Granularity is provider-native: characters on ElevenLabs, words on Cartesia. */
|
||||
export const SpeechTimestamp = Schema.Struct({
|
||||
text: Schema.String,
|
||||
startSeconds: Schema.Number,
|
||||
endSeconds: Schema.Number,
|
||||
}).annotate({ identifier: "Speech.Timestamp" })
|
||||
export type SpeechTimestamp = Schema.Schema.Type<typeof SpeechTimestamp>
|
||||
|
||||
export class SpeechRequest extends Schema.Class<SpeechRequest>("Speech.Request")({
|
||||
model: SpeechModelSchema,
|
||||
text: Schema.String,
|
||||
voice: Schema.optional(SpeechVoice),
|
||||
format: Schema.optional(Schema.String),
|
||||
speed: Schema.optional(Schema.Number),
|
||||
language: Schema.optional(Schema.String),
|
||||
instructions: Schema.optional(Schema.String),
|
||||
timestamps: Schema.optional(Schema.Boolean),
|
||||
providerOptions: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
http: Schema.optional(HttpOptions),
|
||||
}) {
|
||||
declare protected readonly _SpeechRequest: void
|
||||
}
|
||||
|
||||
export type SpeechRequestFor<Options extends SpeechOptions = SpeechOptions> = Omit<
|
||||
SpeechRequest,
|
||||
"model" | "providerOptions"
|
||||
> & {
|
||||
readonly model: SpeechModel<Options>
|
||||
readonly providerOptions?: Options
|
||||
}
|
||||
|
||||
export type SpeechModelOptions<Model> = Model extends SpeechModel<infer Options> ? Options : never
|
||||
|
||||
export type SpeechRequestInput<Model extends SpeechModel = SpeechModel> = Omit<
|
||||
ConstructorParameters<typeof SpeechRequest>[0],
|
||||
"model" | "providerOptions" | "http" | "format"
|
||||
> & {
|
||||
readonly model: Model
|
||||
readonly format?: SpeechFormat
|
||||
readonly providerOptions?: NoInfer<SpeechModelOptions<Model>>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Response and events
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export class SpeechResponse extends Schema.Class<SpeechResponse>("Speech.Response")({
|
||||
/** The complete audio. Headerless PCM carries `info.encoding`, `info.sampleRate`, and `info.channels`. */
|
||||
audio: Media.AssetSchema,
|
||||
timestamps: Schema.optional(Schema.Array(SpeechTimestamp)),
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}) {}
|
||||
|
||||
export const SpeechAudioDeltaEvent = Schema.Struct({
|
||||
type: Schema.tag("audio-delta"),
|
||||
chunk: Schema.Uint8Array,
|
||||
}).annotate({ identifier: "Speech.Event.AudioDelta" })
|
||||
|
||||
export const SpeechTimestampsEvent = Schema.Struct({
|
||||
type: Schema.tag("timestamps"),
|
||||
items: Schema.Array(SpeechTimestamp),
|
||||
}).annotate({ identifier: "Speech.Event.Timestamps" })
|
||||
|
||||
/** `audio` is every `audio-delta` chunk concatenated, so the route holds the whole clip in memory until `finish`. */
|
||||
export const SpeechFinishEvent = Schema.Struct({
|
||||
type: Schema.tag("finish"),
|
||||
audio: Media.AssetSchema,
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}).annotate({ identifier: "Speech.Event.Finish" })
|
||||
|
||||
const speechEventTagged = Schema.Union([SpeechAudioDeltaEvent, SpeechTimestampsEvent, SpeechFinishEvent]).pipe(
|
||||
Schema.toTaggedUnion("type"),
|
||||
)
|
||||
export const SpeechEvent = Object.assign(speechEventTagged, {
|
||||
is: {
|
||||
audioDelta: speechEventTagged.guards["audio-delta"],
|
||||
timestamps: speechEventTagged.guards.timestamps,
|
||||
finish: speechEventTagged.guards.finish,
|
||||
},
|
||||
})
|
||||
export type SpeechEvent = Schema.Schema.Type<typeof speechEventTagged>
|
||||
|
||||
const collectResponse = (events: ReadonlyArray<SpeechEvent>): Effect.Effect<SpeechResponse> => {
|
||||
const finish = events.find(SpeechEvent.is.finish)
|
||||
// Every speech protocol's `finish` emits the terminal event or fails, so a completed stream always has one.
|
||||
if (finish === undefined) return Effect.die(new Error("The speech stream completed without a finish event"))
|
||||
const timestamps = events.filter(SpeechEvent.is.timestamps).flatMap((event) => event.items)
|
||||
return Effect.succeed(
|
||||
new SpeechResponse({
|
||||
audio: finish.audio,
|
||||
timestamps: timestamps.length === 0 ? undefined : timestamps,
|
||||
usage: finish.usage,
|
||||
notices: finish.notices,
|
||||
providerMetadata: finish.providerMetadata,
|
||||
}),
|
||||
)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request-shaped call API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export function request<const Model extends SpeechModel>(
|
||||
input: SpeechRequestInput<Model>,
|
||||
): SpeechRequestFor<SpeechModelOptions<Model>>
|
||||
export function request(input: SpeechRequest): SpeechRequest
|
||||
export function request(input: SpeechRequest | SpeechRequestInput) {
|
||||
if (input instanceof SpeechRequest) return input
|
||||
return new SpeechRequest({
|
||||
...input,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
}
|
||||
|
||||
const requestEffect = (input: SpeechRequest | SpeechRequestInput) => tryRequest(() => request(input))
|
||||
|
||||
export function generate<const Model extends SpeechModel>(
|
||||
input: SpeechRequestInput<Model>,
|
||||
): Effect.Effect<SpeechResponse, AIError, Service>
|
||||
export function generate(input: SpeechRequest): Effect.Effect<SpeechResponse, AIError, Service>
|
||||
export function generate(input: SpeechRequest | SpeechRequestInput) {
|
||||
return requestEffect(input).pipe(Effect.flatMap((request) => SpeechClient.generate(request)))
|
||||
}
|
||||
|
||||
export function stream<const Model extends SpeechModel>(
|
||||
input: SpeechRequestInput<Model>,
|
||||
): Stream.Stream<SpeechEvent, AIError, Service>
|
||||
export function stream(input: SpeechRequest): Stream.Stream<SpeechEvent, AIError, Service>
|
||||
export function stream(input: SpeechRequest | SpeechRequestInput) {
|
||||
return Stream.unwrap(requestEffect(input).pipe(Effect.map((request) => SpeechClient.stream(request))))
|
||||
}
|
||||
|
||||
export const Speech = {
|
||||
request,
|
||||
generate,
|
||||
stream,
|
||||
} as const
|
||||
@@ -0,0 +1,122 @@
|
||||
import { Context, Effect, Layer, Stream } from "effect"
|
||||
import { resultEvents, type AwaitOptions, type Generation } from "./generation.js"
|
||||
import { ProviderShared } from "./protocols/shared.js"
|
||||
import { RequestExecutor } from "./route/executor.js"
|
||||
import type { AIError } from "./schema/index.js"
|
||||
import {
|
||||
responseEvents,
|
||||
type TranscriptionEvent,
|
||||
type TranscriptionModel,
|
||||
type TranscriptionOptions,
|
||||
type TranscriptionRequestFor,
|
||||
type TranscriptionResponse,
|
||||
type TranscriptionRoute,
|
||||
} from "./transcription.js"
|
||||
|
||||
export interface Interface {
|
||||
readonly generate: <Options extends TranscriptionOptions>(
|
||||
request: TranscriptionRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
) => Effect.Effect<TranscriptionResponse, AIError>
|
||||
readonly stream: <Options extends TranscriptionOptions>(
|
||||
request: TranscriptionRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
) => Stream.Stream<TranscriptionEvent, AIError>
|
||||
readonly start: <Options extends TranscriptionOptions>(
|
||||
request: TranscriptionRequestFor<Options>,
|
||||
) => Effect.Effect<Generation<TranscriptionResponse>, AIError>
|
||||
readonly resume: <Options extends TranscriptionOptions>(
|
||||
model: TranscriptionModel<Options>,
|
||||
token: unknown,
|
||||
) => Effect.Effect<Generation<TranscriptionResponse>, AIError>
|
||||
}
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/TranscriptionClient") {}
|
||||
|
||||
export const generate = <Options extends TranscriptionOptions>(
|
||||
request: TranscriptionRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
): Effect.Effect<TranscriptionResponse, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.generate(request, options)
|
||||
})
|
||||
|
||||
export const stream = <Options extends TranscriptionOptions>(
|
||||
request: TranscriptionRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
): Stream.Stream<TranscriptionEvent, AIError, Service> =>
|
||||
Stream.unwrap(
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return client.stream(request, options)
|
||||
}),
|
||||
)
|
||||
|
||||
export const start = <Options extends TranscriptionOptions>(
|
||||
request: TranscriptionRequestFor<Options>,
|
||||
): Effect.Effect<Generation<TranscriptionResponse>, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.start(request)
|
||||
})
|
||||
|
||||
export const resume = <Options extends TranscriptionOptions>(
|
||||
model: TranscriptionModel<Options>,
|
||||
token: unknown,
|
||||
): Effect.Effect<Generation<TranscriptionResponse>, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.resume(model, token)
|
||||
})
|
||||
|
||||
const notQueued = <Options extends TranscriptionOptions>(route: TranscriptionRoute<Options>, operation: string) =>
|
||||
ProviderShared.unsupportedOperation({
|
||||
operation: `transcription.${operation}`,
|
||||
provider: route.provider,
|
||||
route: route.id,
|
||||
message: `${route.provider}/${route.id} is not a queued route; use Transcription.generate or Transcription.stream`,
|
||||
})
|
||||
|
||||
export const layer: Layer.Layer<Service, never, RequestExecutor.Service> = Layer.effect(
|
||||
Service,
|
||||
Effect.gen(function* () {
|
||||
const executor = yield* RequestExecutor.Service
|
||||
const start = <Options extends TranscriptionOptions>(request: TranscriptionRequestFor<Options>) => {
|
||||
const route = request.model.route
|
||||
if (route.kind !== "queued") return Effect.fail(notQueued(route, "start"))
|
||||
return route.start(request, executor.execute)
|
||||
}
|
||||
return Service.of({
|
||||
start,
|
||||
resume: (model, token) => {
|
||||
const route = model.route
|
||||
if (route.kind !== "queued") return Effect.fail(notQueued(route, "resume"))
|
||||
return route.resume(model, token, executor.execute)
|
||||
},
|
||||
generate: (request, options) => {
|
||||
const route = request.model.route
|
||||
if (route.kind !== "queued") return route.generate(request, executor.execute)
|
||||
return start(request).pipe(Effect.flatMap((generation) => generation.await(options)))
|
||||
},
|
||||
stream: (request, options) => {
|
||||
const route = request.model.route
|
||||
if (route.kind === "stream") return route.stream(request, executor.execute)
|
||||
if (route.kind === "queued")
|
||||
return Stream.unwrap(
|
||||
start(request).pipe(Effect.map((generation) => resultEvents(generation, responseEvents, options))),
|
||||
)
|
||||
return Stream.fromIterableEffect(Effect.map(route.generate(request, executor.execute), responseEvents))
|
||||
},
|
||||
})
|
||||
}),
|
||||
)
|
||||
|
||||
export const TranscriptionClient = {
|
||||
Service,
|
||||
layer,
|
||||
generate,
|
||||
stream,
|
||||
start,
|
||||
resume,
|
||||
} as const
|
||||
@@ -0,0 +1,294 @@
|
||||
import { Effect, Schema, Stream } from "effect"
|
||||
import { Generation, ProgressEvent, QueuedEvent, type AwaitOptions } from "./generation.js"
|
||||
import { Media } from "./media.js"
|
||||
import { MediaModel, composeRoute, tryRequest } from "./media-model.js"
|
||||
import { MediaRoute } from "./route/media.js"
|
||||
import type { MediaProtocol } from "./route/media-protocol.js"
|
||||
import { AIError, HttpOptions, MediaUsage, ProviderMetadata } from "./schema/index.js"
|
||||
import { TranscriptionClient, Service } from "./transcription-client.js"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Model
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type TranscriptionOptions = Record<string, unknown>
|
||||
|
||||
export type TranscriptionRoute<Options extends TranscriptionOptions = TranscriptionOptions> =
|
||||
| MediaRoute.Route<TranscriptionRequestFor<Options>, TranscriptionResponse>
|
||||
| MediaRoute.StreamRoute<TranscriptionRequestFor<Options>, TranscriptionEvent, TranscriptionResponse>
|
||||
| MediaRoute.QueuedRoute<TranscriptionRequestFor<Options>, TranscriptionResponse>
|
||||
|
||||
export class TranscriptionModel<Options extends TranscriptionOptions = TranscriptionOptions> extends MediaModel<
|
||||
TranscriptionRoute<Options>,
|
||||
Options
|
||||
> {
|
||||
declare protected readonly _TranscriptionModel: void
|
||||
|
||||
static make<Options extends TranscriptionOptions = TranscriptionOptions>(
|
||||
input: MediaModel.Input<TranscriptionRoute<Options>>,
|
||||
) {
|
||||
return new TranscriptionModel<Options>(input)
|
||||
}
|
||||
|
||||
/** The number of type arguments selects the kind: `<Options>`, `<Options, Frame, State>`, or `<Options, Token>`. */
|
||||
static fromRoute<Options extends TranscriptionOptions>(
|
||||
route: TranscriptionModel.InlineRouteInput<Options>,
|
||||
input: MediaRoute.ModelInput,
|
||||
): TranscriptionModel<Options>
|
||||
static fromRoute<Options extends TranscriptionOptions, Frame, State>(
|
||||
route: TranscriptionModel.StreamRouteInput<Options, Frame, State>,
|
||||
input: MediaRoute.ModelInput,
|
||||
): TranscriptionModel<Options>
|
||||
static fromRoute<Options extends TranscriptionOptions, Token>(
|
||||
route: TranscriptionModel.QueuedRouteInput<Options, Token>,
|
||||
input: MediaRoute.ModelInput,
|
||||
): TranscriptionModel<Options>
|
||||
static fromRoute<Options extends TranscriptionOptions, Frame, State, Token>(
|
||||
route: TranscriptionModel.RouteInput<Options, Frame, State, Token>,
|
||||
input: MediaRoute.ModelInput,
|
||||
) {
|
||||
const composed: TranscriptionRoute<Options> = isStreamInput(route)
|
||||
? composeRoute((composition) => MediaRoute.stream({ ...composition, collect: collectResponse }), route, input)
|
||||
: isQueuedInput(route)
|
||||
? composeRoute(MediaRoute.queued, route, input)
|
||||
: composeRoute(MediaRoute.inline, route, input)
|
||||
return new TranscriptionModel<Options>({
|
||||
id: input.id,
|
||||
provider: route.provider,
|
||||
http: input.http,
|
||||
route: composed,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export namespace TranscriptionModel {
|
||||
export type InlineRouteInput<Options extends TranscriptionOptions = TranscriptionOptions> = MediaModel.RouteInput<
|
||||
TranscriptionRequestFor<Options>,
|
||||
MediaProtocol.Inline<TranscriptionRequestFor<Options>, TranscriptionResponse>
|
||||
>
|
||||
|
||||
export type StreamRouteInput<
|
||||
Options extends TranscriptionOptions = TranscriptionOptions,
|
||||
Frame = unknown,
|
||||
State = unknown,
|
||||
> = MediaModel.RouteInput<
|
||||
MediaProtocol.Addressed<TranscriptionRequestFor<Options>>,
|
||||
MediaProtocol.Streamed<TranscriptionRequestFor<Options>, TranscriptionEvent, Frame, State>
|
||||
>
|
||||
|
||||
export type QueuedRouteInput<
|
||||
Options extends TranscriptionOptions = TranscriptionOptions,
|
||||
Token = unknown,
|
||||
> = MediaModel.RouteInput<
|
||||
TranscriptionRequestFor<Options>,
|
||||
MediaProtocol.Queued<TranscriptionRequestFor<Options>, TranscriptionResponse, Token>
|
||||
>
|
||||
|
||||
export type RouteInput<
|
||||
Options extends TranscriptionOptions = TranscriptionOptions,
|
||||
Frame = unknown,
|
||||
State = unknown,
|
||||
Token = unknown,
|
||||
> = InlineRouteInput<Options> | StreamRouteInput<Options, Frame, State> | QueuedRouteInput<Options, Token>
|
||||
}
|
||||
|
||||
const isStreamInput = <Options extends TranscriptionOptions, Frame, State, Token>(
|
||||
route: TranscriptionModel.RouteInput<Options, Frame, State, Token>,
|
||||
): route is TranscriptionModel.StreamRouteInput<Options, Frame, State> => route.protocol.kind === "stream"
|
||||
|
||||
const isQueuedInput = <Options extends TranscriptionOptions, Frame, State, Token>(
|
||||
route: TranscriptionModel.RouteInput<Options, Frame, State, Token>,
|
||||
): route is TranscriptionModel.QueuedRouteInput<Options, Token> => route.protocol.kind === "queued"
|
||||
|
||||
export const TranscriptionModelSchema = Schema.declare(
|
||||
(value): value is TranscriptionModel => value instanceof TranscriptionModel,
|
||||
{ expected: "Transcription.Model" },
|
||||
)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export const TranscriptionTimestamps = Schema.Literals(["none", "segment", "word"])
|
||||
export type TranscriptionTimestamps = Schema.Schema.Type<typeof TranscriptionTimestamps>
|
||||
|
||||
export class TranscriptionRequest extends Schema.Class<TranscriptionRequest>("Transcription.Request")({
|
||||
model: TranscriptionModelSchema,
|
||||
audio: Media.AssetSchema,
|
||||
language: Schema.optional(Schema.String),
|
||||
prompt: Schema.optional(Schema.String),
|
||||
/** Routes that cannot produce the requested granularity fail typed; routes may return more than asked. */
|
||||
timestamps: Schema.optional(TranscriptionTimestamps),
|
||||
diarize: Schema.optional(Schema.Boolean),
|
||||
speakers: Schema.optional(Schema.Int),
|
||||
providerOptions: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
http: Schema.optional(HttpOptions),
|
||||
}) {
|
||||
declare protected readonly _TranscriptionRequest: void
|
||||
}
|
||||
|
||||
export type TranscriptionRequestFor<Options extends TranscriptionOptions = TranscriptionOptions> = Omit<
|
||||
TranscriptionRequest,
|
||||
"model" | "providerOptions"
|
||||
> & {
|
||||
readonly model: TranscriptionModel<Options>
|
||||
readonly providerOptions?: Options
|
||||
}
|
||||
|
||||
export type TranscriptionModelOptions<Model> = Model extends TranscriptionModel<infer Options> ? Options : never
|
||||
|
||||
export type TranscriptionRequestInput<Model extends TranscriptionModel = TranscriptionModel> = Omit<
|
||||
ConstructorParameters<typeof TranscriptionRequest>[0],
|
||||
"model" | "providerOptions" | "http"
|
||||
> & {
|
||||
readonly model: Model
|
||||
readonly providerOptions?: NoInfer<TranscriptionModelOptions<Model>>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Response and events
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Speaker labels are provider-native (`A`, `0`, `spk:0`, or a known speaker name). */
|
||||
export const TranscriptionSegment = Schema.Struct({
|
||||
text: Schema.String,
|
||||
startSeconds: Schema.Number,
|
||||
endSeconds: Schema.Number,
|
||||
speaker: Schema.optional(Schema.String),
|
||||
}).annotate({ identifier: "Transcription.Segment" })
|
||||
export type TranscriptionSegment = Schema.Schema.Type<typeof TranscriptionSegment>
|
||||
|
||||
export const TranscriptionWord = Schema.Struct({
|
||||
text: Schema.String,
|
||||
startSeconds: Schema.Number,
|
||||
endSeconds: Schema.Number,
|
||||
speaker: Schema.optional(Schema.String),
|
||||
confidence: Schema.optional(Schema.Number),
|
||||
}).annotate({ identifier: "Transcription.Word" })
|
||||
export type TranscriptionWord = Schema.Schema.Type<typeof TranscriptionWord>
|
||||
|
||||
const transcriptFields = {
|
||||
text: Schema.String,
|
||||
segments: Schema.optional(Schema.Array(TranscriptionSegment)),
|
||||
words: Schema.optional(Schema.Array(TranscriptionWord)),
|
||||
/** Provider-native language as detected or echoed (`en`, `english`, `en_us`), lowercased but not normalized. */
|
||||
language: Schema.optional(Schema.String),
|
||||
durationSeconds: Schema.optional(Schema.Number),
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}
|
||||
|
||||
export class TranscriptionResponse extends Schema.Class<TranscriptionResponse>("Transcription.Response")(
|
||||
transcriptFields,
|
||||
) {}
|
||||
|
||||
export const TranscriptionTextDeltaEvent = Schema.Struct({
|
||||
type: Schema.tag("text-delta"),
|
||||
delta: Schema.String,
|
||||
}).annotate({ identifier: "Transcription.Event.TextDelta" })
|
||||
|
||||
export const TranscriptionSegmentEvent = Schema.Struct({
|
||||
type: Schema.tag("segment"),
|
||||
segment: TranscriptionSegment,
|
||||
}).annotate({ identifier: "Transcription.Event.Segment" })
|
||||
|
||||
export const TranscriptionFinishEvent = Schema.Struct({
|
||||
type: Schema.tag("finish"),
|
||||
...transcriptFields,
|
||||
}).annotate({ identifier: "Transcription.Event.Finish" })
|
||||
|
||||
const transcriptionEventTagged = Schema.Union([
|
||||
QueuedEvent,
|
||||
ProgressEvent,
|
||||
TranscriptionTextDeltaEvent,
|
||||
TranscriptionSegmentEvent,
|
||||
TranscriptionFinishEvent,
|
||||
]).pipe(Schema.toTaggedUnion("type"))
|
||||
export const TranscriptionEvent = Object.assign(transcriptionEventTagged, {
|
||||
is: {
|
||||
generationQueued: transcriptionEventTagged.guards["generation-queued"],
|
||||
generationProgress: transcriptionEventTagged.guards["generation-progress"],
|
||||
textDelta: transcriptionEventTagged.guards["text-delta"],
|
||||
segment: transcriptionEventTagged.guards.segment,
|
||||
finish: transcriptionEventTagged.guards.finish,
|
||||
},
|
||||
})
|
||||
export type TranscriptionEvent = Schema.Schema.Type<typeof transcriptionEventTagged>
|
||||
|
||||
export const responseEvents = (response: TranscriptionResponse): ReadonlyArray<TranscriptionEvent> => [
|
||||
TranscriptionFinishEvent.make({ ...response }),
|
||||
]
|
||||
|
||||
const collectResponse = (events: ReadonlyArray<TranscriptionEvent>): Effect.Effect<TranscriptionResponse> => {
|
||||
const finish = events.find(TranscriptionEvent.is.finish)
|
||||
// Every transcription protocol's `finish` emits the terminal event or fails, so a completed stream always has one.
|
||||
if (finish === undefined) return Effect.die(new Error("The transcription stream completed without a finish event"))
|
||||
const { type: _type, ...transcript } = finish
|
||||
return Effect.succeed(new TranscriptionResponse(transcript))
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request-shaped call API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export function request<const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model>,
|
||||
): TranscriptionRequestFor<TranscriptionModelOptions<Model>>
|
||||
export function request(input: TranscriptionRequest): TranscriptionRequest
|
||||
export function request(input: TranscriptionRequest | TranscriptionRequestInput) {
|
||||
if (input instanceof TranscriptionRequest) return input
|
||||
return new TranscriptionRequest({
|
||||
...input,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
}
|
||||
|
||||
const requestEffect = (input: TranscriptionRequest | TranscriptionRequestInput) => tryRequest(() => request(input))
|
||||
|
||||
export function generate<const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model>,
|
||||
options?: AwaitOptions,
|
||||
): Effect.Effect<TranscriptionResponse, AIError, Service>
|
||||
export function generate(
|
||||
input: TranscriptionRequest,
|
||||
options?: AwaitOptions,
|
||||
): Effect.Effect<TranscriptionResponse, AIError, Service>
|
||||
export function generate(input: TranscriptionRequest | TranscriptionRequestInput, options?: AwaitOptions) {
|
||||
return requestEffect(input).pipe(Effect.flatMap((request) => TranscriptionClient.generate(request, options)))
|
||||
}
|
||||
|
||||
export function stream<const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model>,
|
||||
options?: AwaitOptions,
|
||||
): Stream.Stream<TranscriptionEvent, AIError, Service>
|
||||
export function stream(
|
||||
input: TranscriptionRequest,
|
||||
options?: AwaitOptions,
|
||||
): Stream.Stream<TranscriptionEvent, AIError, Service>
|
||||
export function stream(input: TranscriptionRequest | TranscriptionRequestInput, options?: AwaitOptions) {
|
||||
return Stream.unwrap(requestEffect(input).pipe(Effect.map((request) => TranscriptionClient.stream(request, options))))
|
||||
}
|
||||
|
||||
/** Inline and streaming routes fail with `UnsupportedOperation`. */
|
||||
export function start<const Model extends TranscriptionModel>(
|
||||
input: TranscriptionRequestInput<Model>,
|
||||
): Effect.Effect<Generation<TranscriptionResponse>, AIError, Service>
|
||||
export function start(input: TranscriptionRequest): Effect.Effect<Generation<TranscriptionResponse>, AIError, Service>
|
||||
export function start(input: TranscriptionRequest | TranscriptionRequestInput) {
|
||||
return requestEffect(input).pipe(Effect.flatMap((request) => TranscriptionClient.start(request)))
|
||||
}
|
||||
|
||||
export const resume = <Options extends TranscriptionOptions>(
|
||||
model: TranscriptionModel<Options>,
|
||||
token: unknown,
|
||||
): Effect.Effect<Generation<TranscriptionResponse>, AIError, Service> => TranscriptionClient.resume(model, token)
|
||||
|
||||
export const Transcription = {
|
||||
request,
|
||||
generate,
|
||||
stream,
|
||||
start,
|
||||
resume,
|
||||
} as const
|
||||
@@ -0,0 +1,9 @@
|
||||
export const concatBytes = (chunks: ReadonlyArray<Uint8Array>) => {
|
||||
if (chunks.length === 1) return chunks[0]
|
||||
const bytes = new Uint8Array(chunks.reduce((total, chunk) => total + chunk.length, 0))
|
||||
chunks.reduce((offset, chunk) => {
|
||||
bytes.set(chunk, offset)
|
||||
return offset + chunk.length
|
||||
}, 0)
|
||||
return bytes
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
const ascii = (bytes: Uint8Array, start: number, end: number) => String.fromCharCode(...bytes.subarray(start, end))
|
||||
|
||||
const startsWith = (bytes: Uint8Array, prefix: ReadonlyArray<number>) =>
|
||||
bytes.length >= prefix.length && prefix.every((value, index) => bytes[index] === value)
|
||||
|
||||
/**
|
||||
* Sniff a media type from leading magic bytes. Covers the containers media routes commonly return; anything else is
|
||||
* `undefined` so callers can fall back to a provider-declared type or `application/octet-stream`.
|
||||
*/
|
||||
export const detectMediaType = (bytes: Uint8Array): string | undefined => {
|
||||
if (startsWith(bytes, [0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a])) return "image/png"
|
||||
if (startsWith(bytes, [0xff, 0xd8, 0xff])) return "image/jpeg"
|
||||
if (startsWith(bytes, [0x47, 0x49, 0x46, 0x38])) return "image/gif"
|
||||
if (bytes.length >= 12 && ascii(bytes, 0, 4) === "RIFF") {
|
||||
const riffType = ascii(bytes, 8, 12)
|
||||
if (riffType === "WEBP") return "image/webp"
|
||||
if (riffType === "WAVE") return "audio/wav"
|
||||
}
|
||||
if (startsWith(bytes, [0x25, 0x50, 0x44, 0x46])) return "application/pdf"
|
||||
if (bytes.length >= 12 && ascii(bytes, 4, 8) === "ftyp") return "video/mp4"
|
||||
if (startsWith(bytes, [0x1a, 0x45, 0xdf, 0xa3])) return "video/webm"
|
||||
if (startsWith(bytes, [0x49, 0x44, 0x33])) return "audio/mpeg"
|
||||
// An 11-bit frame sync; layer bits `00` mark AAC ADTS, any other layer is MPEG audio.
|
||||
if (bytes.length >= 2 && bytes[0] === 0xff && (bytes[1] & 0xe0) === 0xe0)
|
||||
return (bytes[1] & 0x06) === 0 ? "audio/aac" : "audio/mpeg"
|
||||
if (startsWith(bytes, [0x4f, 0x67, 0x67, 0x53])) return "audio/ogg"
|
||||
if (startsWith(bytes, [0x66, 0x4c, 0x61, 0x43])) return "audio/flac"
|
||||
return undefined
|
||||
}
|
||||
|
||||
const EXTENSIONS: Readonly<Record<string, string>> = {
|
||||
png: "image/png",
|
||||
jpg: "image/jpeg",
|
||||
jpeg: "image/jpeg",
|
||||
gif: "image/gif",
|
||||
webp: "image/webp",
|
||||
pdf: "application/pdf",
|
||||
mp4: "video/mp4",
|
||||
webm: "video/webm",
|
||||
mp3: "audio/mpeg",
|
||||
m4a: "audio/mp4",
|
||||
wav: "audio/wav",
|
||||
ogg: "audio/ogg",
|
||||
flac: "audio/flac",
|
||||
aac: "audio/aac",
|
||||
txt: "text/plain",
|
||||
md: "text/markdown",
|
||||
csv: "text/csv",
|
||||
}
|
||||
|
||||
export const extensionMediaType = (path: string): string | undefined =>
|
||||
EXTENSIONS[path.slice(path.lastIndexOf(".") + 1).toLowerCase()]
|
||||
|
||||
const EXTENSION_ALIASES: Readonly<Record<string, string>> = {
|
||||
"audio/mp3": "mp3",
|
||||
"audio/m4a": "m4a",
|
||||
"audio/x-m4a": "m4a",
|
||||
"audio/webm": "webm",
|
||||
"audio/wave": "wav",
|
||||
"audio/x-wav": "wav",
|
||||
"audio/x-flac": "flac",
|
||||
}
|
||||
|
||||
export const mediaTypeExtension = (mediaType: string): string | undefined => {
|
||||
const type = mediaType.split(";", 1)[0].trim().toLowerCase()
|
||||
return EXTENSION_ALIASES[type] ?? Object.entries(EXTENSIONS).find(([, known]) => known === type)?.[0]
|
||||
}
|
||||
@@ -1,9 +1,11 @@
|
||||
import { Media } from "../media.js"
|
||||
import { isRecord } from "./record.js"
|
||||
|
||||
export const sanitizeSurrogates = <T>(value: T): T => {
|
||||
if (typeof value === "string") return value.toWellFormed() as T
|
||||
if (Array.isArray(value)) return value.map(sanitizeSurrogates) as T
|
||||
if (value instanceof Uint8Array || value instanceof Error) return value
|
||||
// Media assets carry binary or base64 payloads and a lazy byte cache; flattening them into a record would drop both.
|
||||
if (value instanceof Uint8Array || value instanceof Error || value instanceof Media.Asset) return value
|
||||
if (isRecord(value))
|
||||
return Object.fromEntries(
|
||||
Object.entries(value).map(([key, entry]) => [key.toWellFormed(), sanitizeSurrogates(entry)]),
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
import { Context, Effect, Layer, Stream } from "effect"
|
||||
import { resultEvents, type AwaitOptions, type Generation } from "./generation.js"
|
||||
import { RequestExecutor } from "./route/executor.js"
|
||||
import type { AIError } from "./schema/index.js"
|
||||
import {
|
||||
responseEvents,
|
||||
type VideoEvent,
|
||||
type VideoModel,
|
||||
type VideoOptions,
|
||||
type VideoRequestFor,
|
||||
type VideoResponse,
|
||||
} from "./video.js"
|
||||
|
||||
export interface Interface {
|
||||
readonly start: <Options extends VideoOptions>(
|
||||
request: VideoRequestFor<Options>,
|
||||
) => Effect.Effect<Generation<VideoResponse>, AIError>
|
||||
readonly resume: <Options extends VideoOptions>(
|
||||
model: VideoModel<Options>,
|
||||
token: unknown,
|
||||
) => Effect.Effect<Generation<VideoResponse>, AIError>
|
||||
readonly generate: <Options extends VideoOptions>(
|
||||
request: VideoRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
) => Effect.Effect<VideoResponse, AIError>
|
||||
readonly stream: <Options extends VideoOptions>(
|
||||
request: VideoRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
) => Stream.Stream<VideoEvent, AIError>
|
||||
}
|
||||
|
||||
export class Service extends Context.Service<Service, Interface>()("@opencode/VideoClient") {}
|
||||
|
||||
export const start = <Options extends VideoOptions>(
|
||||
request: VideoRequestFor<Options>,
|
||||
): Effect.Effect<Generation<VideoResponse>, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.start(request)
|
||||
})
|
||||
|
||||
export const resume = <Options extends VideoOptions>(
|
||||
model: VideoModel<Options>,
|
||||
token: unknown,
|
||||
): Effect.Effect<Generation<VideoResponse>, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.resume(model, token)
|
||||
})
|
||||
|
||||
export const generate = <Options extends VideoOptions>(
|
||||
request: VideoRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
): Effect.Effect<VideoResponse, AIError, Service> =>
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return yield* client.generate(request, options)
|
||||
})
|
||||
|
||||
export const stream = <Options extends VideoOptions>(
|
||||
request: VideoRequestFor<Options>,
|
||||
options?: AwaitOptions,
|
||||
): Stream.Stream<VideoEvent, AIError, Service> =>
|
||||
Stream.unwrap(
|
||||
Effect.gen(function* () {
|
||||
const client = yield* Service
|
||||
return client.stream(request, options)
|
||||
}),
|
||||
)
|
||||
|
||||
export const layer: Layer.Layer<Service, never, RequestExecutor.Service> = Layer.effect(
|
||||
Service,
|
||||
Effect.gen(function* () {
|
||||
const executor = yield* RequestExecutor.Service
|
||||
const start = <Options extends VideoOptions>(request: VideoRequestFor<Options>) =>
|
||||
request.model.route.start(request, executor.execute)
|
||||
return Service.of({
|
||||
start,
|
||||
resume: (model, token) => model.route.resume(model, token, executor.execute),
|
||||
generate: (request, options) => start(request).pipe(Effect.flatMap((generation) => generation.await(options))),
|
||||
stream: (request, options) =>
|
||||
Stream.unwrap(
|
||||
start(request).pipe(Effect.map((generation) => resultEvents(generation, responseEvents, options))),
|
||||
),
|
||||
})
|
||||
}),
|
||||
)
|
||||
|
||||
export const VideoClient = {
|
||||
Service,
|
||||
layer,
|
||||
start,
|
||||
resume,
|
||||
generate,
|
||||
stream,
|
||||
} as const
|
||||
@@ -0,0 +1,218 @@
|
||||
import { Effect, Schema, Stream } from "effect"
|
||||
import { Generation, ProgressEvent, QueuedEvent, type AwaitOptions } from "./generation.js"
|
||||
import { Media } from "./media.js"
|
||||
import { MediaModel, composeRoute, tryRequest } from "./media-model.js"
|
||||
import { MediaRoute } from "./route/media.js"
|
||||
import type { MediaProtocol } from "./route/media-protocol.js"
|
||||
import { AIError, HttpOptions, MediaUsage, ProviderMetadata } from "./schema/index.js"
|
||||
import { VideoClient, Service } from "./video-client.js"
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Model
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type VideoOptions = Record<string, unknown>
|
||||
|
||||
export type VideoRoute<Options extends VideoOptions = VideoOptions> = MediaRoute.QueuedRoute<
|
||||
VideoRequestFor<Options>,
|
||||
VideoResponse
|
||||
>
|
||||
|
||||
export class VideoModel<Options extends VideoOptions = VideoOptions> extends MediaModel<VideoRoute<Options>, Options> {
|
||||
declare protected readonly _VideoModel: void
|
||||
|
||||
static make<Options extends VideoOptions = VideoOptions>(input: MediaModel.Input<VideoRoute<Options>>) {
|
||||
return new VideoModel<Options>(input)
|
||||
}
|
||||
|
||||
/** Compose a queued video protocol with its canonical start path into a model for one deployment. */
|
||||
static fromRoute<Options extends VideoOptions = VideoOptions, Token = unknown>(
|
||||
route: VideoModel.RouteInput<Options, Token>,
|
||||
input: MediaRoute.ModelInput,
|
||||
) {
|
||||
return new VideoModel<Options>({
|
||||
id: input.id,
|
||||
provider: route.provider,
|
||||
http: input.http,
|
||||
route: composeRoute(MediaRoute.queued, route, input),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
export namespace VideoModel {
|
||||
export type RouteInput<Options extends VideoOptions = VideoOptions, Token = unknown> = MediaModel.RouteInput<
|
||||
VideoRequestFor<Options>,
|
||||
MediaProtocol.Queued<VideoRequestFor<Options>, VideoResponse, Token>
|
||||
>
|
||||
}
|
||||
|
||||
export const VideoModelSchema = Schema.declare((value): value is VideoModel => value instanceof VideoModel, {
|
||||
expected: "Video.Model",
|
||||
})
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export type VideoAspectRatio = Media.AspectRatio
|
||||
export const VideoAspectRatio = Media.AspectRatio
|
||||
|
||||
export type VideoResolution = "480p" | "720p" | "1080p" | "4k" | (string & {})
|
||||
|
||||
/** Pinned frames. Routes that accept only a first frame fail typed when `last` is present. */
|
||||
export const VideoFrames = Schema.Struct({
|
||||
first: Schema.optional(Media.AssetSchema),
|
||||
last: Schema.optional(Media.AssetSchema),
|
||||
}).annotate({ identifier: "Video.Frames" })
|
||||
export type VideoFrames = Schema.Schema.Type<typeof VideoFrames>
|
||||
|
||||
export class VideoRequest extends Schema.Class<VideoRequest>("Video.Request")({
|
||||
model: VideoModelSchema,
|
||||
prompt: Schema.String,
|
||||
frames: Schema.optional(VideoFrames),
|
||||
/** Style or subject references that guide the output without pinning a frame. */
|
||||
references: Schema.optional(Schema.Array(Media.AssetSchema)),
|
||||
/** Source video for edit or extension routes. */
|
||||
video: Schema.optional(Media.AssetSchema),
|
||||
durationSeconds: Schema.optional(Schema.Number),
|
||||
aspectRatio: Schema.optional(VideoAspectRatio),
|
||||
resolution: Schema.optional(Schema.String),
|
||||
/** Whether to generate an audio track; routes whose audio is always on fail typed on `false`. */
|
||||
audio: Schema.optional(Schema.Boolean),
|
||||
n: Schema.optional(Schema.Int),
|
||||
seed: Schema.optional(Schema.Number),
|
||||
negativePrompt: Schema.optional(Schema.String),
|
||||
providerOptions: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)),
|
||||
http: Schema.optional(HttpOptions),
|
||||
}) {
|
||||
declare protected readonly _VideoRequest: void
|
||||
}
|
||||
|
||||
export type VideoRequestFor<Options extends VideoOptions = VideoOptions> = Omit<
|
||||
VideoRequest,
|
||||
"model" | "providerOptions"
|
||||
> & {
|
||||
readonly model: VideoModel<Options>
|
||||
readonly providerOptions?: Options
|
||||
}
|
||||
|
||||
export type VideoModelOptions<Model> = Model extends VideoModel<infer Options> ? Options : never
|
||||
|
||||
export type VideoRequestInput<Model extends VideoModel = VideoModel> = Omit<
|
||||
ConstructorParameters<typeof VideoRequest>[0],
|
||||
"model" | "providerOptions" | "http" | "resolution"
|
||||
> & {
|
||||
readonly model: Model
|
||||
readonly resolution?: VideoResolution
|
||||
readonly providerOptions?: NoInfer<VideoModelOptions<Model>>
|
||||
readonly http?: HttpOptions.Input
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Response and events
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export class VideoResponse extends Schema.Class<VideoResponse>("Video.Response")({
|
||||
videos: Schema.Array(Media.AssetSchema),
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}) {
|
||||
get video() {
|
||||
return this.videos[0]
|
||||
}
|
||||
}
|
||||
|
||||
export const VideoOutputEvent = Schema.Struct({
|
||||
type: Schema.tag("video"),
|
||||
index: Schema.Number,
|
||||
video: Media.AssetSchema,
|
||||
}).annotate({ identifier: "Video.Event.Video" })
|
||||
|
||||
export const VideoFinishEvent = Schema.Struct({
|
||||
type: Schema.tag("finish"),
|
||||
usage: Schema.optional(MediaUsage),
|
||||
notices: Schema.optional(Schema.Array(Media.Notice)),
|
||||
providerMetadata: Schema.optional(ProviderMetadata),
|
||||
}).annotate({ identifier: "Video.Event.Finish" })
|
||||
|
||||
const videoEventTagged = Schema.Union([QueuedEvent, ProgressEvent, VideoOutputEvent, VideoFinishEvent]).pipe(
|
||||
Schema.toTaggedUnion("type"),
|
||||
)
|
||||
export const VideoEvent = Object.assign(videoEventTagged, {
|
||||
is: {
|
||||
generationQueued: videoEventTagged.guards["generation-queued"],
|
||||
generationProgress: videoEventTagged.guards["generation-progress"],
|
||||
video: videoEventTagged.guards.video,
|
||||
finish: videoEventTagged.guards.finish,
|
||||
},
|
||||
})
|
||||
export type VideoEvent = Schema.Schema.Type<typeof videoEventTagged>
|
||||
|
||||
/** A completed response expanded into the streaming event shape. */
|
||||
export const responseEvents = (response: VideoResponse): ReadonlyArray<VideoEvent> => [
|
||||
...response.videos.map((video, index) => VideoOutputEvent.make({ index, video })),
|
||||
VideoFinishEvent.make({
|
||||
usage: response.usage,
|
||||
notices: response.notices,
|
||||
providerMetadata: response.providerMetadata,
|
||||
}),
|
||||
]
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Request-shaped call API
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export function request<const Model extends VideoModel>(
|
||||
input: VideoRequestInput<Model>,
|
||||
): VideoRequestFor<VideoModelOptions<Model>>
|
||||
export function request(input: VideoRequest): VideoRequest
|
||||
export function request(input: VideoRequest | VideoRequestInput) {
|
||||
if (input instanceof VideoRequest) return input
|
||||
return new VideoRequest({
|
||||
...input,
|
||||
http: input.http === undefined ? undefined : HttpOptions.make(input.http),
|
||||
})
|
||||
}
|
||||
|
||||
const requestEffect = (input: VideoRequest | VideoRequestInput) => tryRequest(() => request(input))
|
||||
|
||||
export function start<const Model extends VideoModel>(
|
||||
input: VideoRequestInput<Model>,
|
||||
): Effect.Effect<Generation<VideoResponse>, AIError, Service>
|
||||
export function start(input: VideoRequest): Effect.Effect<Generation<VideoResponse>, AIError, Service>
|
||||
export function start(input: VideoRequest | VideoRequestInput) {
|
||||
return requestEffect(input).pipe(Effect.flatMap((request) => VideoClient.start(request)))
|
||||
}
|
||||
|
||||
export function generate<const Model extends VideoModel>(
|
||||
input: VideoRequestInput<Model>,
|
||||
options?: AwaitOptions,
|
||||
): Effect.Effect<VideoResponse, AIError, Service>
|
||||
export function generate(input: VideoRequest, options?: AwaitOptions): Effect.Effect<VideoResponse, AIError, Service>
|
||||
export function generate(input: VideoRequest | VideoRequestInput, options?: AwaitOptions) {
|
||||
return requestEffect(input).pipe(Effect.flatMap((request) => VideoClient.generate(request, options)))
|
||||
}
|
||||
|
||||
/** Rebuild a generation handle from a persisted `Generation.token`, refreshing its status once. */
|
||||
export const resume = <Options extends VideoOptions>(
|
||||
model: VideoModel<Options>,
|
||||
token: unknown,
|
||||
): Effect.Effect<Generation<VideoResponse>, AIError, Service> => VideoClient.resume(model, token)
|
||||
|
||||
export function stream<const Model extends VideoModel>(
|
||||
input: VideoRequestInput<Model>,
|
||||
options?: AwaitOptions,
|
||||
): Stream.Stream<VideoEvent, AIError, Service>
|
||||
export function stream(input: VideoRequest, options?: AwaitOptions): Stream.Stream<VideoEvent, AIError, Service>
|
||||
export function stream(input: VideoRequest | VideoRequestInput, options?: AwaitOptions) {
|
||||
return Stream.unwrap(requestEffect(input).pipe(Effect.map((request) => VideoClient.stream(request, options))))
|
||||
}
|
||||
|
||||
export const Video = {
|
||||
request,
|
||||
start,
|
||||
generate,
|
||||
resume,
|
||||
stream,
|
||||
} as const
|
||||
@@ -87,9 +87,9 @@ testEffect(fixedResponse("")).effect(
|
||||
prompt: "hello",
|
||||
})
|
||||
expect(LLMClient.canCompact(request)).toBe(false)
|
||||
const error = yield* LLMClient.compact(
|
||||
request as unknown as Parameters<typeof LLMClient.compact>[0],
|
||||
).pipe(Effect.flip)
|
||||
const error = yield* LLMClient.compact(request as unknown as Parameters<typeof LLMClient.compact>[0]).pipe(
|
||||
Effect.flip,
|
||||
)
|
||||
expect(error.reason._tag).toBe("UnsupportedOperation")
|
||||
expect(error.message).toContain("does not support explicit compaction")
|
||||
if (error.reason._tag === "UnsupportedOperation") {
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import {
|
||||
LLM,
|
||||
Media,
|
||||
Message,
|
||||
ToolCallPart,
|
||||
ToolDefinition,
|
||||
@@ -59,7 +60,7 @@ export function continuationRequest(input: {
|
||||
|
||||
if (features.has("user-text")) firstUser.push({ type: "text", text: "What is shown here?" })
|
||||
if (features.has("user-image"))
|
||||
firstUser.push({ type: "media", mediaType: "image/png", data: input.image ?? "AAECAw==" })
|
||||
firstUser.push({ type: "media", media: Media.base64(input.image ?? "AAECAw==", "image/png") })
|
||||
if (firstUser.length > 0) messages.push(Message.user(firstUser))
|
||||
|
||||
if (features.has("assistant-reasoning"))
|
||||
|
||||
@@ -0,0 +1,302 @@
|
||||
import { describe, expect } from "bun:test"
|
||||
import { Effect, Layer } from "effect"
|
||||
import { HttpClientRequest } from "effect/unstable/http"
|
||||
import { Evaluation, EvaluationClient } from "../src/experimental.js"
|
||||
import { OpenCodeZen, OpenRouter, TypeSafeAI, VercelAIGateway } from "../src/providers.js"
|
||||
import { it } from "./lib/effect.js"
|
||||
import { dynamicResponse } from "./lib/http.js"
|
||||
|
||||
describe("experimental Evaluation", () => {
|
||||
it.effect("evaluates typed questions through System One", () =>
|
||||
Effect.gen(function* () {
|
||||
const response = yield* Evaluation.run({
|
||||
model: TypeSafeAI.configure({
|
||||
apiKey: "test",
|
||||
baseURL: "https://typesafe.test/v1/",
|
||||
headers: { "x-default": "yes" },
|
||||
http: { body: { deployment: "test" }, query: { api: "v1" } },
|
||||
}).experimental.evaluation("jev-latest"),
|
||||
state: { ticket: "Please refund the duplicate charge." },
|
||||
questions: {
|
||||
department: {
|
||||
type: "choice",
|
||||
instructions: "Which team should handle this?",
|
||||
criteria: { billing: "Payments and refunds", technical: "Bugs and outages" },
|
||||
},
|
||||
urgency: {
|
||||
type: "score",
|
||||
instructions: "How urgent is this?",
|
||||
criteria: ["Can wait", "Needs attention", "Blocking"],
|
||||
},
|
||||
refund: { type: "boolean", instructions: "Is the customer asking for a refund?" },
|
||||
},
|
||||
options: { trace: { enabled: true } },
|
||||
http: { body: { request_metadata: "value" }, headers: { "x-request": "yes" }, query: { trace: "1" } },
|
||||
})
|
||||
|
||||
expect(response.model).toBe("jev-1.13.0")
|
||||
expect(response.answers.department).toEqual({
|
||||
type: "choice",
|
||||
choice: "billing",
|
||||
probabilities: { billing: 0.9, technical: 0.1 },
|
||||
})
|
||||
expect(response.answers.urgency).toEqual({
|
||||
type: "score",
|
||||
score: 1.2,
|
||||
probabilities: { "0": 0, "1": 0.8, "2": 0.2 },
|
||||
})
|
||||
expect(response.answers.refund).toEqual({ type: "boolean", probability: 0.97 })
|
||||
expect(response.usage?.totalTokens).toBe(36)
|
||||
expect(response.providerMetadata).toEqual({
|
||||
typesafe: {
|
||||
confidence: { department: 0.8, urgency: 0.6 },
|
||||
legend: { urgency: { "0": "Can wait", "1": "Needs attention", "2": "Blocking" } },
|
||||
},
|
||||
})
|
||||
}).pipe(
|
||||
Effect.provide(
|
||||
EvaluationClient.layer.pipe(
|
||||
Layer.provide(
|
||||
dynamicResponse((input) =>
|
||||
Effect.gen(function* () {
|
||||
const request = yield* HttpClientRequest.toWeb(input.request).pipe(Effect.orDie)
|
||||
expect(request.url).toBe("https://typesafe.test/v1/systemone?api=v1&trace=1")
|
||||
expect(request.headers.get("authorization")).toBe("Bearer test")
|
||||
expect(request.headers.get("x-default")).toBe("yes")
|
||||
expect(request.headers.get("x-request")).toBe("yes")
|
||||
expect(JSON.parse(input.text)).toEqual({
|
||||
deployment: "test",
|
||||
request_metadata: "value",
|
||||
trace: { enabled: true },
|
||||
model: "jev-latest",
|
||||
state: { ticket: "Please refund the duplicate charge." },
|
||||
questions: {
|
||||
department: {
|
||||
type: "choice",
|
||||
instructions: "Which team should handle this?",
|
||||
criteria: { billing: "Payments and refunds", technical: "Bugs and outages" },
|
||||
},
|
||||
urgency: {
|
||||
type: "score",
|
||||
instructions: "How urgent is this?",
|
||||
criteria: ["Can wait", "Needs attention", "Blocking"],
|
||||
},
|
||||
refund: { type: "noul", instructions: "Is the customer asking for a refund?" },
|
||||
},
|
||||
})
|
||||
return input.respond(
|
||||
JSON.stringify({
|
||||
model: "jev-1.13.0",
|
||||
answers: {
|
||||
department: {
|
||||
type: "choice",
|
||||
choice: "billing",
|
||||
probabilities: { billing: 0.9, technical: 0.1 },
|
||||
confidence: 0.8,
|
||||
},
|
||||
urgency: {
|
||||
type: "score",
|
||||
score: 1.2,
|
||||
probabilities: { "0": 0, "1": 0.8, "2": 0.2 },
|
||||
legend: { "0": "Can wait", "1": "Needs attention", "2": "Blocking" },
|
||||
confidence: 0.6,
|
||||
},
|
||||
refund: { type: "noul", noul: 0.97 },
|
||||
},
|
||||
usage: { input_tokens: 30, output_tokens: 6 },
|
||||
}),
|
||||
{ headers: { "content-type": "application/json" } },
|
||||
)
|
||||
}),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
it.effect("configures the OpenCode Zen System One endpoint", () =>
|
||||
Evaluation.run({
|
||||
model: OpenCodeZen.configure({ apiKey: "zen-key", baseURL: "https://zen.test/v1" }).experimental.evaluation(
|
||||
"jev-1.13",
|
||||
),
|
||||
state: "hello",
|
||||
questions: { greeting: { type: "boolean", instructions: "Is this a greeting?" } },
|
||||
}).pipe(
|
||||
Effect.tap((response) =>
|
||||
Effect.sync(() => {
|
||||
expect(response.answers.greeting.probability).toBe(0.99)
|
||||
expect(response.usage?.providerMetadata).toEqual({
|
||||
opencode: { input_tokens: 10, output_tokens: 2 },
|
||||
})
|
||||
}),
|
||||
),
|
||||
Effect.provide(
|
||||
EvaluationClient.layer.pipe(
|
||||
Layer.provide(
|
||||
dynamicResponse((input) => {
|
||||
expect(input.request.url).toBe("https://zen.test/v1/systemone")
|
||||
expect(input.request.headers.authorization).toBe("Bearer zen-key")
|
||||
return Effect.succeed(
|
||||
input.respond(
|
||||
JSON.stringify({
|
||||
model: "jev-1.13.0",
|
||||
answers: { greeting: { type: "noul", noul: 0.99 } },
|
||||
usage: { input_tokens: 10, output_tokens: 2 },
|
||||
}),
|
||||
{ headers: { "content-type": "application/json" } },
|
||||
),
|
||||
)
|
||||
}),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
it.effect("evaluates through OpenRouter System One", () =>
|
||||
Evaluation.run({
|
||||
model: OpenRouter.configure({
|
||||
apiKey: "openrouter-key",
|
||||
baseURL: "https://openrouter.test/api/v1",
|
||||
}).experimental.evaluation("typesafe/jev-1.13"),
|
||||
state: "refund",
|
||||
questions: { refund: { type: "boolean", instructions: "Is a refund requested?" } },
|
||||
options: { user: "user-1", session_id: "session-1" },
|
||||
}).pipe(
|
||||
Effect.tap((response) =>
|
||||
Effect.sync(() => {
|
||||
expect(response.answers.refund.probability).toBe(0.98)
|
||||
expect(response.providerMetadata?.openrouter).toMatchObject({ responseId: "gen-1", provider: "TypeSafe" })
|
||||
expect(response.usage?.providerMetadata?.openrouter).toEqual({
|
||||
input_tokens: 10,
|
||||
output_tokens: 2,
|
||||
cost: 0.0001,
|
||||
})
|
||||
}),
|
||||
),
|
||||
Effect.provide(
|
||||
EvaluationClient.layer.pipe(
|
||||
Layer.provide(
|
||||
dynamicResponse((input) => {
|
||||
expect(input.request.url).toBe("https://openrouter.test/api/v1/systemone")
|
||||
expect(input.request.headers.authorization).toBe("Bearer openrouter-key")
|
||||
expect(JSON.parse(input.text)).toMatchObject({
|
||||
model: "typesafe/jev-1.13",
|
||||
user: "user-1",
|
||||
session_id: "session-1",
|
||||
questions: { refund: { type: "noul" } },
|
||||
})
|
||||
return Effect.succeed(
|
||||
input.respond(
|
||||
JSON.stringify({
|
||||
id: "gen-1",
|
||||
model: "typesafe/jev-1.13-20260917",
|
||||
provider: "TypeSafe",
|
||||
answers: { refund: { type: "noul", noul: 0.98 } },
|
||||
usage: { input_tokens: 10, output_tokens: 2, cost: 0.0001 },
|
||||
}),
|
||||
{ headers: { "content-type": "application/json" } },
|
||||
),
|
||||
)
|
||||
}),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
it.effect("evaluates through Vercel AI Gateway", () =>
|
||||
Evaluation.run({
|
||||
model: VercelAIGateway.configure({
|
||||
apiKey: "gateway-key",
|
||||
baseURL: "https://gateway.test/v1/",
|
||||
}).experimental.evaluation("typesafe-ai/jev"),
|
||||
state: "refund",
|
||||
questions: { refund: { type: "boolean", instructions: "Is a refund requested?" } },
|
||||
options: { gateway: { zeroDataRetention: true, only: ["typesafe-ai"] } },
|
||||
}).pipe(
|
||||
Effect.tap((response) =>
|
||||
Effect.sync(() => {
|
||||
expect(response.answers.refund.probability).toBe(0.98)
|
||||
expect(response.usage?.totalTokens).toBe(12)
|
||||
expect(response.providerMetadata?.gateway).toMatchObject({ generationId: "gen-1", cost: "0.0001" })
|
||||
}),
|
||||
),
|
||||
Effect.provide(
|
||||
EvaluationClient.layer.pipe(
|
||||
Layer.provide(
|
||||
dynamicResponse((input) => {
|
||||
expect(input.request.url).toBe("https://gateway.test/v1/evaluate")
|
||||
expect(input.request.headers.authorization).toBe("Bearer gateway-key")
|
||||
expect(JSON.parse(input.text)).toEqual({
|
||||
model: "typesafe-ai/jev",
|
||||
state: "refund",
|
||||
questions: { refund: { type: "boolean", instructions: "Is a refund requested?" } },
|
||||
providerOptions: { gateway: { zeroDataRetention: true, only: ["typesafe-ai"] } },
|
||||
})
|
||||
return Effect.succeed(
|
||||
input.respond(
|
||||
JSON.stringify({
|
||||
model: "typesafe-ai/jev",
|
||||
answers: { refund: { type: "boolean", probability: 0.98 } },
|
||||
usage: { inputTokens: 10, outputTokens: 2 },
|
||||
providerMetadata: { gateway: { generationId: "gen-1", cost: "0.0001" } },
|
||||
}),
|
||||
{ headers: { "content-type": "application/json" } },
|
||||
),
|
||||
)
|
||||
}),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
it.effect("rejects answers that do not match their questions", () =>
|
||||
Evaluation.run({
|
||||
model: VercelAIGateway.configure({
|
||||
apiKey: "gateway-key",
|
||||
baseURL: "https://gateway.test/v1",
|
||||
}).experimental.evaluation("typesafe-ai/jev"),
|
||||
state: "refund",
|
||||
questions: { refund: { type: "boolean", instructions: "Is a refund requested?" } },
|
||||
}).pipe(
|
||||
Effect.flip,
|
||||
Effect.tap((error) => Effect.sync(() => expect(error.reason._tag).toBe("InvalidProviderOutput"))),
|
||||
Effect.provide(
|
||||
EvaluationClient.layer.pipe(
|
||||
Layer.provide(
|
||||
dynamicResponse((input) =>
|
||||
Effect.succeed(
|
||||
input.respond(
|
||||
JSON.stringify({
|
||||
answers: { refund: { type: "choice", choice: "yes", probabilities: { yes: 1 } } },
|
||||
}),
|
||||
{ headers: { "content-type": "application/json" } },
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
it.effect("rejects malformed questions before network I/O", () =>
|
||||
Effect.gen(function* () {
|
||||
const error = yield* Evaluation.run({
|
||||
model: TypeSafeAI.experimental.evaluation("jev-latest"),
|
||||
state: "hello",
|
||||
questions: { score: { type: "score", instructions: "How much?", criteria: ["only"] } },
|
||||
}).pipe(Effect.flip)
|
||||
expect(error.reason._tag).toBe("InvalidRequest")
|
||||
}).pipe(
|
||||
Effect.provide(
|
||||
EvaluationClient.layer.pipe(
|
||||
Layer.provide(dynamicResponse(() => Effect.die("invalid evaluation reached the network"))),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
})
|
||||
@@ -0,0 +1,88 @@
|
||||
import { Effect } from "effect"
|
||||
import { Evaluation, EvaluationClient, EvaluationModel, type EvaluationRoute } from "../src/experimental.js"
|
||||
import type { Service } from "../src/experimental/evaluation-client.js"
|
||||
import { OpenCodeZen, OpenRouter, TypeSafeAI, VercelAIGateway } from "../src/providers.js"
|
||||
|
||||
type Requirements<T> = T extends Effect.Effect<infer _A, infer _E, infer R> ? R : never
|
||||
type Success<T> = T extends Effect.Effect<infer A, infer _E, infer _R> ? A : never
|
||||
type Equal<A, B> = [A, B] extends [B, A] ? true : false
|
||||
type Assert<T extends true> = T
|
||||
|
||||
const model = TypeSafeAI.configure({ apiKey: "test" }).experimental.evaluation("jev-latest")
|
||||
const request = Evaluation.request({
|
||||
model,
|
||||
state: { ticket: "refund" },
|
||||
questions: {
|
||||
topic: {
|
||||
type: "choice",
|
||||
instructions: "Which team?",
|
||||
criteria: { billing: null, support: { includes: ["help"] } },
|
||||
},
|
||||
severity: { type: "score", instructions: "How severe?", criteria: ["Low", "High"] },
|
||||
refund: { type: "boolean", instructions: "Refund?" },
|
||||
},
|
||||
})
|
||||
|
||||
const result = EvaluationClient.evaluate(request)
|
||||
type Result = Success<typeof result>
|
||||
type Choice = Assert<Equal<Result["answers"]["topic"]["choice"], "billing" | "support">>
|
||||
type ClientRequirements = Assert<Equal<Requirements<typeof result>, Service>>
|
||||
void (true satisfies Choice)
|
||||
void (true satisfies ClientRequirements)
|
||||
|
||||
Effect.gen(function* () {
|
||||
const response = yield* Evaluation.run({
|
||||
model: OpenCodeZen.experimental.evaluation("jev-1.13"),
|
||||
state: ["hello"],
|
||||
questions: { greeting: { type: "boolean", instructions: "Greeting?" } },
|
||||
})
|
||||
response.answers.greeting.probability satisfies number
|
||||
// @ts-expect-error Boolean answers do not contain a selected choice.
|
||||
response.answers.greeting.choice
|
||||
// @ts-expect-error Unknown question IDs are not exposed.
|
||||
response.answers.missing
|
||||
})
|
||||
|
||||
declare const route: EvaluationRoute<{ readonly temperature?: number }>
|
||||
const custom = EvaluationModel.make({ id: "custom", provider: "custom", route })
|
||||
Evaluation.run({
|
||||
model: custom,
|
||||
state: "hello",
|
||||
questions: { ok: { type: "boolean", instructions: "OK?" } },
|
||||
options: { temperature: 0.5 },
|
||||
})
|
||||
// @ts-expect-error Selected evaluation models retain their request option types.
|
||||
Evaluation.run({
|
||||
model: custom,
|
||||
state: "hello",
|
||||
questions: { ok: { type: "boolean", instructions: "OK?" } },
|
||||
options: { temperature: "high" },
|
||||
})
|
||||
|
||||
Evaluation.run({
|
||||
model: OpenRouter.experimental.evaluation("typesafe/jev-1.13"),
|
||||
state: "hello",
|
||||
questions: { ok: { type: "boolean", instructions: "OK?" } },
|
||||
options: { provider: { zdr: true }, session_id: "session-1", user: "user-1" },
|
||||
})
|
||||
// @ts-expect-error OpenRouter session IDs are strings.
|
||||
Evaluation.run({
|
||||
model: OpenRouter.experimental.evaluation("typesafe/jev-1.13"),
|
||||
state: "hello",
|
||||
questions: { ok: { type: "boolean", instructions: "OK?" } },
|
||||
options: { session_id: 1 },
|
||||
})
|
||||
|
||||
Evaluation.run({
|
||||
model: VercelAIGateway.experimental.evaluation("typesafe-ai/jev"),
|
||||
state: "hello",
|
||||
questions: { ok: { type: "boolean", instructions: "OK?" } },
|
||||
options: { gateway: { zeroDataRetention: true, only: ["typesafe-ai"] } },
|
||||
})
|
||||
// @ts-expect-error Vercel zero-data-retention controls are boolean.
|
||||
Evaluation.run({
|
||||
model: VercelAIGateway.experimental.evaluation("typesafe-ai/jev"),
|
||||
state: "hello",
|
||||
questions: { ok: { type: "boolean", instructions: "OK?" } },
|
||||
options: { gateway: { zeroDataRetention: "yes" } },
|
||||
})
|
||||
@@ -1,16 +1,40 @@
|
||||
import { describe, expect, test } from "bun:test"
|
||||
import { AIError, ImageInput, LanguageModel, LLM, LLMClient, Provider } from "@opencode/ai"
|
||||
import {
|
||||
AIError,
|
||||
Generation,
|
||||
Image,
|
||||
LanguageModel,
|
||||
LLM,
|
||||
LLMClient,
|
||||
Media,
|
||||
Provider,
|
||||
Speech,
|
||||
SpeechClient,
|
||||
SpeechEvent,
|
||||
Video,
|
||||
VideoClient,
|
||||
} from "@opencode/ai"
|
||||
import { Route, Protocol, WebSocketTransport } from "@opencode/ai/route"
|
||||
import { Provider as ProviderSubpath } from "@opencode/ai/provider"
|
||||
import {
|
||||
AssemblyAI,
|
||||
Baseten,
|
||||
Cartesia,
|
||||
CloudflareAIGateway,
|
||||
CloudflareWorkersAI,
|
||||
Deepgram,
|
||||
DeepSeek,
|
||||
ElevenLabs,
|
||||
Fal,
|
||||
Fireworks,
|
||||
Google,
|
||||
OpenCodeZen,
|
||||
OpenAI,
|
||||
OpenAICompatible,
|
||||
OpenRouter,
|
||||
Runway,
|
||||
TypeSafeAI,
|
||||
VercelAIGateway,
|
||||
XAI,
|
||||
} from "@opencode/ai/providers"
|
||||
import {
|
||||
@@ -23,6 +47,7 @@ import {
|
||||
} from "@opencode/ai/protocols"
|
||||
import * as AnthropicMessages from "@opencode/ai/protocols/anthropic-messages"
|
||||
import { TestLLM } from "@opencode/ai/testing"
|
||||
import { Evaluation, EvaluationClient } from "@opencode/ai/experimental"
|
||||
|
||||
describe("public exports", () => {
|
||||
test("root exposes app-facing runtime APIs", () => {
|
||||
@@ -31,12 +56,24 @@ describe("public exports", () => {
|
||||
expect(LLMClient.layer).toBeDefined()
|
||||
expect(AIError).toBeFunction()
|
||||
expect(LanguageModel.make).toBeFunction()
|
||||
expect(ImageInput.bytes).toBeFunction()
|
||||
expect(Media.bytes).toBeFunction()
|
||||
expect(Image.generate).toBeFunction()
|
||||
expect(Video.start).toBeFunction()
|
||||
expect(Video.resume).toBeFunction()
|
||||
expect(VideoClient.layer).toBeDefined()
|
||||
expect(Speech.generate).toBeFunction()
|
||||
expect(Speech.stream).toBeFunction()
|
||||
expect(SpeechClient.layer).toBeDefined()
|
||||
expect(SpeechEvent.is.audioDelta).toBeFunction()
|
||||
expect(Generation).toBeFunction()
|
||||
expect(Provider.make).toBeFunction()
|
||||
expect(ProviderSubpath.make).toBe(Provider.make)
|
||||
expect(TestLLM.layer).toBeFunction()
|
||||
expect(TestLLM.testLayer).toBeFunction()
|
||||
expect(TestLLM.Test.of).toBeFunction()
|
||||
expect(Evaluation.run).toBeFunction()
|
||||
expect(EvaluationClient.layer).toBeDefined()
|
||||
expect(EvaluationClient.fetchLayer).toBeDefined()
|
||||
})
|
||||
|
||||
test("route barrel exposes route-authoring APIs", () => {
|
||||
@@ -66,11 +103,30 @@ describe("public exports", () => {
|
||||
expect(CloudflareWorkersAI.configure).toBeFunction()
|
||||
expect(CloudflareWorkersAI.configure({ accountId: "fixture", apiKey: "fixture" }).model).toBeFunction()
|
||||
expect(OpenRouter.model).toBeFunction()
|
||||
expect(OpenRouter.experimental.evaluation).toBeFunction()
|
||||
expect(TypeSafeAI.experimental.evaluation).toBeFunction()
|
||||
expect(OpenCodeZen.experimental.evaluation).toBeFunction()
|
||||
expect(VercelAIGateway.experimental.evaluation).toBeFunction()
|
||||
expect(XAI.model).toBeFunction()
|
||||
expect(XAI.provider.responses).toBe(XAI.responses)
|
||||
expect(XAI.provider.chat).toBe(XAI.chat)
|
||||
expect(XAI.configure({ apiKey: "fixture" }).responses("grok-4.3").route.id).toBe("openai-responses")
|
||||
expect(XAI.configure({ apiKey: "fixture" }).chat("grok-4.3").route.id).toBe("openai-compatible-chat")
|
||||
expect(XAI.configure({ apiKey: "fixture" }).video("grok-imagine-video-1.5").route.id).toBe("xai-video")
|
||||
expect(Fal.configure({ apiKey: "fixture" }).video("fal-ai/veo3.1").route.id).toBe("fal-video")
|
||||
expect(Runway.configure({ apiKey: "fixture" }).video("gen4.5").route.id).toBe("runway-video")
|
||||
expect(Runway.provider.video).toBe(Runway.video)
|
||||
expect(OpenAI.configure({ apiKey: "fixture" }).speech("gpt-4o-mini-tts").route.id).toBe("openai-speech")
|
||||
expect(Google.configure({ apiKey: "fixture" }).speech("gemini-2.5-flash-preview-tts").route.id).toBe(
|
||||
"google-speech",
|
||||
)
|
||||
expect(ElevenLabs.configure({ apiKey: "fixture" }).speech("eleven_flash_v2_5").route.id).toBe("elevenlabs-speech")
|
||||
expect(Cartesia.configure({ apiKey: "fixture" }).speech("sonic-3").route.id).toBe("cartesia-speech")
|
||||
expect(Deepgram.configure({ apiKey: "fixture" }).speech("aura-2-thalia-en").route.id).toBe("deepgram-speech")
|
||||
expect(OpenAI.configure({ apiKey: "fixture" }).transcription("gpt-transcribe").route.kind).toBe("stream")
|
||||
expect(Google.configure({ apiKey: "fixture" }).transcription("gemini-3.5-transcribe").route.kind).toBe("stream")
|
||||
expect(Deepgram.configure({ apiKey: "fixture" }).transcription("nova-3").route.kind).toBe("inline")
|
||||
expect(AssemblyAI.configure({ apiKey: "fixture" }).transcription("universal-3-5-pro").route.kind).toBe("queued")
|
||||
})
|
||||
|
||||
test("protocol barrels expose supported low-level routes", () => {
|
||||
|
||||
BIN
Binary file not shown.
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user