feat(models): add Inkling and Fireworks DeepSeek V4 Flash 0731, delist gateway GLM-5.1 (#220)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-08-06 20:08:20 +08:00
committed by GitHub
parent a59dc504c5
commit e07106a924
5 changed files with 67 additions and 38 deletions
+46 -30
View File
@@ -17,10 +17,12 @@
*
* Scope: excludes deepseek-chat / deepseek-reasoner legacy aliases that AgentHub cannot
* auto-route (deprecated 2026-07-24), glm-5v-turbo (image input unsupported by AgentHub's GLM
* client), non-chat models (embedding / image generation / TTS), and Bedrock. Direct-vendor
* ids are auto-routed by AgentHub and leave client_type unset; the five gateway groups
* (OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go) can't be
* auto-routed, so they set `client_type: "openai"` and inline their preset base URL.
* client), the OpenRouter z-ai/glm-5.1 and SiliconFlow Pro/zai-org/GLM-5.1 gateway listings
* (delisted 2026-08-06; the Z.AI direct glm-5.1 remains), non-chat models (embedding / image
* generation / TTS), and Bedrock. Direct-vendor ids are auto-routed by AgentHub and leave
* client_type unset; the five gateway groups (OpenRouter, Fireworks AI, SiliconFlow, Qwen
* Token Plan, Qwen Pay-As-You-Go) can't be auto-routed, so they set `client_type: "openai"`
* and inline their preset base URL.
*
* This file imports no Node built-ins (type-only imports only), so it can be bundled directly
* for the browser.
@@ -222,9 +224,9 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
// -- OpenRouter (gateway: OpenAI-compatible protocol, preset base URL). Prices re-read in
// one pass on 2026-08-03 from the models API (/api/v1/models): cache_read stores the
// published input_cache_read (falling back to the input price for the few rows without
// one — qwen3.6-35b-a3b and the :free rows); cache_write stores input_cache_write only
// when it is a genuine per-token write premium (the Anthropic, GPT and qwen3.8-max rows,
// 1.25x input) —
// one — qwen3.6-35b-a3b, thinkingmachines/inkling and the :free rows); cache_write stores
// input_cache_write only when it is a genuine per-token write premium (the Anthropic, GPT
// and qwen3.8-max rows, 1.25x input) —
// Gemini's field is an hourly cache-STORAGE rate, not a per-token price, so those rows
// keep the input price — and otherwise also carries the input price. The :free tier and
// the openrouter/free Free Models Router store a genuine $0 price (not "unknown"), so
@@ -503,6 +505,20 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// Thinking Machines Lab's Inkling (released 2026-07-14): multimodal (image + audio
// input). Specs and pricing from its OpenRouter page (2026-08-06), which publishes no
// cached-input price, so cache_read repeats the input price (no discount assumed; see
// the block comment above).
modelId: "thinkingmachines/inkling",
displayName: "Inkling",
provider: "openrouter",
contextWindow: 1000000,
pricing: usd(0.95, 0.95, 4.05),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "x-ai/grok-4.5",
displayName: "Grok 4.5",
@@ -533,19 +549,19 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "z-ai/glm-5.1",
displayName: "GLM-5.1",
provider: "openrouter",
contextWindow: 204800,
pricing: usd(0.1794, 0.966, 3.036),
supportsVision: false,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
// -- Fireworks AI (gateway, standard serverless USD pricing: cached input / uncached
// input / output from each model's page; API ids use the accounts/fireworks/models/<slug>
// form) --
{
modelId: "accounts/fireworks/models/deepseek-v4-flash-0731",
displayName: "DeepSeek V4 Flash 0731",
provider: "fireworks",
contextWindow: 1000000,
pricing: usd(0.028, 0.14, 0.28),
supportsVision: false,
clientType: "openai",
baseUrl: FIREWORKS_BASE_URL,
},
{
modelId: "accounts/fireworks/models/deepseek-v4-flash",
displayName: "DeepSeek V4 Flash",
@@ -576,6 +592,18 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: FIREWORKS_BASE_URL,
},
{
// Thinking Machines Lab's Inkling (released 2026-07-14): multimodal (image + audio
// input); specs and serverless pricing from its Fireworks model page (2026-08-06).
modelId: "accounts/fireworks/models/inkling",
displayName: "Inkling",
provider: "fireworks",
contextWindow: 1000000,
pricing: usd(0.17, 1, 4.05),
supportsVision: true,
clientType: "openai",
baseUrl: FIREWORKS_BASE_URL,
},
{
modelId: "accounts/fireworks/models/kimi-k3",
displayName: "Kimi K3",
@@ -649,9 +677,7 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
},
// The Pro/ and Qwen/ entries below were unpriced until 2026-08-03 (SiliconFlow's price
// list sits behind an authenticated console); prices below are its official CNY list
// prices. GLM-5.1 bills in two input-length tiers ([0, 32k) and [32k, +inf) for hit/input/
// output alike); the catalog stores one number per bucket, so these rows keep the LOWER
// tier — treat its cost as a floor for long-context use.
// prices.
{
modelId: "Pro/moonshotai/Kimi-K2.6",
displayName: "Kimi K2.6",
@@ -662,16 +688,6 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "Pro/zai-org/GLM-5.1",
displayName: "GLM-5.1",
provider: "siliconflow",
contextWindow: 200000,
pricing: cny(1.3, 6, 24),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
// No cache-hit price on the list, so cache_read carries the input price.
modelId: "Qwen/Qwen3.6-35B-A3B",