feat(models): add Inkling and Fireworks DeepSeek V4 Flash 0731, delist gateway GLM-5.1 (#220)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -17,10 +17,12 @@
|
||||
*
|
||||
* Scope: excludes deepseek-chat / deepseek-reasoner legacy aliases that AgentHub cannot
|
||||
* auto-route (deprecated 2026-07-24), glm-5v-turbo (image input unsupported by AgentHub's GLM
|
||||
* client), non-chat models (embedding / image generation / TTS), and Bedrock. Direct-vendor
|
||||
* ids are auto-routed by AgentHub and leave client_type unset; the five gateway groups
|
||||
* (OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go) can't be
|
||||
* auto-routed, so they set `client_type: "openai"` and inline their preset base URL.
|
||||
* client), the OpenRouter z-ai/glm-5.1 and SiliconFlow Pro/zai-org/GLM-5.1 gateway listings
|
||||
* (delisted 2026-08-06; the Z.AI direct glm-5.1 remains), non-chat models (embedding / image
|
||||
* generation / TTS), and Bedrock. Direct-vendor ids are auto-routed by AgentHub and leave
|
||||
* client_type unset; the five gateway groups (OpenRouter, Fireworks AI, SiliconFlow, Qwen
|
||||
* Token Plan, Qwen Pay-As-You-Go) can't be auto-routed, so they set `client_type: "openai"`
|
||||
* and inline their preset base URL.
|
||||
*
|
||||
* This file imports no Node built-ins (type-only imports only), so it can be bundled directly
|
||||
* for the browser.
|
||||
@@ -222,9 +224,9 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
// -- OpenRouter (gateway: OpenAI-compatible protocol, preset base URL). Prices re-read in
|
||||
// one pass on 2026-08-03 from the models API (/api/v1/models): cache_read stores the
|
||||
// published input_cache_read (falling back to the input price for the few rows without
|
||||
// one — qwen3.6-35b-a3b and the :free rows); cache_write stores input_cache_write only
|
||||
// when it is a genuine per-token write premium (the Anthropic, GPT and qwen3.8-max rows,
|
||||
// 1.25x input) —
|
||||
// one — qwen3.6-35b-a3b, thinkingmachines/inkling and the :free rows); cache_write stores
|
||||
// input_cache_write only when it is a genuine per-token write premium (the Anthropic, GPT
|
||||
// and qwen3.8-max rows, 1.25x input) —
|
||||
// Gemini's field is an hourly cache-STORAGE rate, not a per-token price, so those rows
|
||||
// keep the input price — and otherwise also carries the input price. The :free tier and
|
||||
// the openrouter/free Free Models Router store a genuine $0 price (not "unknown"), so
|
||||
@@ -503,6 +505,20 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
// Thinking Machines Lab's Inkling (released 2026-07-14): multimodal (image + audio
|
||||
// input). Specs and pricing from its OpenRouter page (2026-08-06), which publishes no
|
||||
// cached-input price, so cache_read repeats the input price (no discount assumed; see
|
||||
// the block comment above).
|
||||
modelId: "thinkingmachines/inkling",
|
||||
displayName: "Inkling",
|
||||
provider: "openrouter",
|
||||
contextWindow: 1000000,
|
||||
pricing: usd(0.95, 0.95, 4.05),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "x-ai/grok-4.5",
|
||||
displayName: "Grok 4.5",
|
||||
@@ -533,19 +549,19 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "z-ai/glm-5.1",
|
||||
displayName: "GLM-5.1",
|
||||
provider: "openrouter",
|
||||
contextWindow: 204800,
|
||||
pricing: usd(0.1794, 0.966, 3.036),
|
||||
supportsVision: false,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
// -- Fireworks AI (gateway, standard serverless USD pricing: cached input / uncached
|
||||
// input / output from each model's page; API ids use the accounts/fireworks/models/<slug>
|
||||
// form) --
|
||||
{
|
||||
modelId: "accounts/fireworks/models/deepseek-v4-flash-0731",
|
||||
displayName: "DeepSeek V4 Flash 0731",
|
||||
provider: "fireworks",
|
||||
contextWindow: 1000000,
|
||||
pricing: usd(0.028, 0.14, 0.28),
|
||||
supportsVision: false,
|
||||
clientType: "openai",
|
||||
baseUrl: FIREWORKS_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "accounts/fireworks/models/deepseek-v4-flash",
|
||||
displayName: "DeepSeek V4 Flash",
|
||||
@@ -576,6 +592,18 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: FIREWORKS_BASE_URL,
|
||||
},
|
||||
{
|
||||
// Thinking Machines Lab's Inkling (released 2026-07-14): multimodal (image + audio
|
||||
// input); specs and serverless pricing from its Fireworks model page (2026-08-06).
|
||||
modelId: "accounts/fireworks/models/inkling",
|
||||
displayName: "Inkling",
|
||||
provider: "fireworks",
|
||||
contextWindow: 1000000,
|
||||
pricing: usd(0.17, 1, 4.05),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: FIREWORKS_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "accounts/fireworks/models/kimi-k3",
|
||||
displayName: "Kimi K3",
|
||||
@@ -649,9 +677,7 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
},
|
||||
// The Pro/ and Qwen/ entries below were unpriced until 2026-08-03 (SiliconFlow's price
|
||||
// list sits behind an authenticated console); prices below are its official CNY list
|
||||
// prices. GLM-5.1 bills in two input-length tiers ([0, 32k) and [32k, +inf) for hit/input/
|
||||
// output alike); the catalog stores one number per bucket, so these rows keep the LOWER
|
||||
// tier — treat its cost as a floor for long-context use.
|
||||
// prices.
|
||||
{
|
||||
modelId: "Pro/moonshotai/Kimi-K2.6",
|
||||
displayName: "Kimi K2.6",
|
||||
@@ -662,16 +688,6 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: SILICONFLOW_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "Pro/zai-org/GLM-5.1",
|
||||
displayName: "GLM-5.1",
|
||||
provider: "siliconflow",
|
||||
contextWindow: 200000,
|
||||
pricing: cny(1.3, 6, 24),
|
||||
supportsVision: false,
|
||||
clientType: "openai",
|
||||
baseUrl: SILICONFLOW_BASE_URL,
|
||||
},
|
||||
{
|
||||
// No cache-hit price on the list, so cache_read carries the input price.
|
||||
modelId: "Qwen/Qwen3.6-35B-A3B",
|
||||
|
||||
Reference in New Issue
Block a user