Compare commits

...
4 changed files with 71 additions and 8 deletions

No files matched your search

+29
View File
@@ -46,6 +46,20 @@ jobs:
- uses: ./.github/actions/setup-bun
# The bundled snapshot is the boot-time catalog floor on a cold cache. Refresh it once per
# release and share the same file with every job that bundles core.
- name: Refresh models.dev snapshot
# A models.dev outage must not block a release; the committed snapshot ships instead.
continue-on-error: true
working-directory: packages/core
run: bun run update-models-snapshot
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: models-dev-snapshot
path: packages/core/src/models-dev/snapshot.txt
if-no-files-found: error
- name: Deploy update service
if: github.ref_name == 'v2'
working-directory: services/update
@@ -89,6 +103,11 @@ jobs:
with:
fetch-tags: true
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: models-dev-snapshot
path: packages/core/src/models-dev
- uses: ./.github/actions/setup-bun
with:
bun-version: 1.4.2
@@ -203,6 +222,11 @@ jobs:
steps:
- uses: actions/checkout@f43a0e5ff2bd294095638e18286ca9a3d1956744 # v3.6.0
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: models-dev-snapshot
path: packages/core/src/models-dev
- uses: ./.github/actions/setup-bun
- name: Build app archive
@@ -550,6 +574,11 @@ jobs:
steps:
- uses: actions/checkout@f43a0e5ff2bd294095638e18286ca9a3d1956744 # v3.6.0
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: models-dev-snapshot
path: packages/core/src/models-dev
- uses: ./.github/actions/setup-bun
- name: Login to GitHub Container Registry
+1 -1
View File
@@ -1 +1 @@
{"zhipuai":{"id":"zhipuai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/paas/v4","name":"Zhipu AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":5,"output":22,"cache_read":1.2,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]}Line truncated
{"subconscious":{"id":"subconscious","env":["SUBCONSCIOUS_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.subconscious.dev/v1","name":"Subconscious","doc":"https://docs.subconscious.dev","models":{"subconscious/glm-5.2":{"id":"subconscious/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"subconscious/tim-qwen3.6-27b":{"id":"subconscious/tim-qwen3.6-27b","name":"TIM-Qwen3.6 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":5000},"cost":{"input":0.3,"output":3,"cache_read":0.15}}}},"tokengo":{"id":"tokengo","env":["TOKENGO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokengo.com/v1","name":"TokenGo","doc":"https://www.tokengo.com/docs","models":{"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":2.65,"cache_read":0.2}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.098,"output":0.196,"cache_read":0.028}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.2174,"output":0.326,"cache_read":0.06}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.19,"output":0.71,"cache_read":0.06}},"moonshLine truncated
+10 -7
View File
@@ -57,16 +57,19 @@ export const ModelsDevPlugin = define({
})
}
})
const apply = (data: readonly ModelsDev.Snapshot[]) => {
loaded.data = snapshots(data)
return ctx.integration.reload().pipe(Effect.andThen(ctx.provider.reload()))
}
yield* bus.subscribe(ModelsDev.Event.Refreshed).pipe(
Stream.runForEach(() =>
modelsDev.get().pipe(
Effect.tap((data) => Effect.sync(() => (loaded.data = snapshots(data)))),
Effect.andThen(ctx.integration.reload()),
Effect.andThen(ctx.provider.reload()),
),
),
Stream.runForEach(() => modelsDev.get().pipe(Effect.flatMap(apply))),
Effect.forkScoped({ startImmediately: true }),
)
// A refresh that landed between the initial read and the subscription above published
// Refreshed to nobody here. On a cold cache that read served the bundled snapshot, so
// re-read now instead of waiting for the next TTL refresh.
const latest = yield* modelsDev.get()
if (snapshots(latest) !== loaded.data) yield* apply(latest)
}),
})
@@ -290,6 +290,37 @@ describe("ModelsDevPlugin", () => {
}),
)
isolated.effect("adopts a refresh that completes between its initial read and its subscription", () =>
Effect.gen(function* () {
const bundled = richSnapshot("Acme Bundled")
const fresh = richSnapshot("Acme Fresh")
const current = { snapshot: bundled.snapshot }
const location = yield* owner
// Cold cache: the first read serves the bundled snapshot, and the boot-time
// ModelsDev.refresh() lands right after it, before the plugin subscribes.
const source = ModelsDev.Service.of({
get: () =>
Effect.gen(function* () {
const data = current.snapshot
if (data !== bundled.snapshot) return data
current.snapshot = fresh.snapshot
yield* location.bus.publish(ModelsDev.Event.Refreshed, {})
return data
}),
refresh: () => Effect.void,
})
yield* ModelsDevPlugin.effect(location.host).pipe(
Effect.provideService(ModelsDev.Service, source),
Effect.provideContext(location.context),
)
yield* TestClock.adjust("500 millis")
yield* TestClock.adjust("500 millis")
yield* TestClock.adjust("500 millis")
expect(required(yield* location.providers.get(bundled.providerID)).name).toBe("Acme Fresh")
}),
)
real.effect("keeps the retained definition unchanged across model replay", () =>
Effect.gen(function* () {
const providers = yield* Provider.Service