feat: AgentHub 0.4.1 and a model-catalog refresh across every provider group (#43)
Co-authored-by: Alice <alice@prismshadow.com> Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -82,16 +82,15 @@ Four Skill groups ship in the box ([docs](https://penguin.ooo/docs/skills)); Age
|
||||
| Model | Providers |
|
||||
| ---------------- | -------------------------------------------------------------------------------- |
|
||||
| DeepSeek V4 | DeepSeek, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan |
|
||||
| Kimi K3 | OpenRouter, Qwen Pay-As-You-Go |
|
||||
| Kimi K2.6 | Moonshot AI |
|
||||
| Kimi K3 | Moonshot AI, OpenRouter, Qwen Pay-As-You-Go |
|
||||
| GLM 5.2 | Z.AI, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go |
|
||||
| Hunyuan 3 | OpenRouter |
|
||||
| Qwen 3.8 Max | Qwen Token Plan (preview) |
|
||||
| GPT 5.5 | OpenAI, OpenRouter |
|
||||
| Gemini 3.5 Flash | Google Gemini, OpenRouter |
|
||||
| Claude Opus 4.8 | Anthropic, OpenRouter |
|
||||
| GPT 5.6 | OpenRouter |
|
||||
| Gemini 3.6 Flash | Google Gemini, OpenRouter |
|
||||
| Claude 5 | Anthropic, OpenRouter |
|
||||
|
||||
Any OpenAI-protocol endpoint is supported: pick a preset above, or point a custom endpoint at any of the 1000+ online and local models.
|
||||
Each family's latest generation only — the app's **Models** page lists every built-in preset, and any OpenAI-protocol endpoint works too: pick a preset, or point a custom endpoint at any of the 1000+ online and local models.
|
||||
|
||||
## Requirements
|
||||
|
||||
|
||||
+5
-6
@@ -82,16 +82,15 @@ https://github.com/user-attachments/assets/aec49ae9-b743-467b-b247-37bedfeaa36e
|
||||
| 模型 | 可用供应商 |
|
||||
| ---------------- | -------------------------------------------------------------------------------- |
|
||||
| DeepSeek V4 | DeepSeek, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan |
|
||||
| Kimi K3 | OpenRouter, Qwen Pay-As-You-Go |
|
||||
| Kimi K2.6 | Moonshot AI |
|
||||
| Kimi K3 | Moonshot AI, OpenRouter, Qwen Pay-As-You-Go |
|
||||
| GLM 5.2 | Z.AI, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go |
|
||||
| Hunyuan 3 | OpenRouter |
|
||||
| Qwen 3.8 Max | Qwen Token Plan(预览) |
|
||||
| GPT 5.5 | OpenAI, OpenRouter |
|
||||
| Gemini 3.5 Flash | Google Gemini, OpenRouter |
|
||||
| Claude Opus 4.8 | Anthropic, OpenRouter |
|
||||
| GPT 5.6 | OpenRouter |
|
||||
| Gemini 3.6 Flash | Google Gemini, OpenRouter |
|
||||
| Claude 5 | Anthropic, OpenRouter |
|
||||
|
||||
只要是 OpenAI 协议的端点都可以接入:从上表选择预置,或用自定义端点连接 1000+ 在线与本地模型。
|
||||
上表每个系列只列最新一代,完整预置清单请在应用的**模型**页查看;只要是 OpenAI 协议的端点都可以接入:选择预置,或用自定义端点连接 1000+ 在线与本地模型。
|
||||
|
||||
## 系统需求
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@
|
||||
"penguin": "tsx src/index.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
"@prismshadow/agenthub": "^0.4.0",
|
||||
"@prismshadow/agenthub": "^0.4.1",
|
||||
"@prismshadow/penguin-core": "workspace:*",
|
||||
"@prismshadow/penguin-server": "workspace:*",
|
||||
"@prismshadow/penguin-skills": "workspace:*",
|
||||
|
||||
@@ -36,7 +36,7 @@
|
||||
"build": "tsup"
|
||||
},
|
||||
"dependencies": {
|
||||
"@prismshadow/agenthub": "^0.4.0",
|
||||
"@prismshadow/agenthub": "^0.4.1",
|
||||
"@prismshadow/penguin-skills": "workspace:*",
|
||||
"smol-toml": "^1.3.0",
|
||||
"yaml": "^2.5.0"
|
||||
|
||||
@@ -222,7 +222,12 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
// 2026-07-20 list no cache pricing on their OpenRouter pages, so cache_read carries the
|
||||
// standard input price; their :free tier stores a genuine $0 price (not "unknown"), so
|
||||
// costs correctly compute to 0. GPT models are uniformly vision-capable (OpenAI
|
||||
// product-line policy) even where the gateway page omits the modality. --
|
||||
// product-line policy) even where the gateway page omits the modality.
|
||||
// Two cache_read conventions coexist in this block: rows whose upstream publishes a
|
||||
// cache-hit price store that real price (the google/gemini-3.6-flash and
|
||||
// google/gemini-3.5-flash-lite entries below), while the 2026-07-20 rows still repeat the
|
||||
// input price. Several of those older rows do have a published cache price upstream and
|
||||
// should be re-read in one pass; until then treat their cache_read as an upper bound. --
|
||||
{
|
||||
modelId: "anthropic/claude-fable-5",
|
||||
displayName: "Claude Fable 5",
|
||||
@@ -283,16 +288,45 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
// Unlike the older OpenRouter entries above, upstream **does** publish a cache-hit price
|
||||
// for the Gemini rows (2026-07-22: $0.15/mtok here, agreed by the OpenRouter models API
|
||||
// and AgentHub's own supported-model registry), so cache_read stores the real discounted
|
||||
// price rather than repeating the input price: cache_read is billed as its own bucket in
|
||||
// the cost center, and an input-priced cache_read overstates cache-heavy spend 10x.
|
||||
// cache_write repeats the input price (no separate per-token cache-write fee), matching
|
||||
// the direct-vendor Gemini rows below.
|
||||
modelId: "google/gemini-3.6-flash",
|
||||
displayName: "Gemini 3.6 Flash",
|
||||
provider: "openrouter",
|
||||
contextWindow: 1048576,
|
||||
pricing: usd(0.15, 1.5, 7.5),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "google/gemini-3.5-flash",
|
||||
displayName: "Gemini 3.5 Flash",
|
||||
provider: "openrouter",
|
||||
contextWindow: 1000000,
|
||||
contextWindow: 1048576,
|
||||
pricing: usd(1.5, 1.5, 9),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
// Same published-cache-price convention as gemini-3.6-flash above (2026-07-22: $0.03/mtok
|
||||
// cache hit, $0.30 input, $2.50 output).
|
||||
modelId: "google/gemini-3.5-flash-lite",
|
||||
displayName: "Gemini 3.5 Flash-Lite",
|
||||
provider: "openrouter",
|
||||
contextWindow: 1048576,
|
||||
pricing: usd(0.03, 0.3, 2.5),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
// No official separate cache price published: cache_read uses the standard input price (no discount assumed).
|
||||
modelId: "minimax/minimax-m3",
|
||||
@@ -314,6 +348,16 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "moonshotai/kimi-k2.6",
|
||||
displayName: "Kimi K2.6",
|
||||
provider: "openrouter",
|
||||
contextWindow: 262144,
|
||||
pricing: usd(0.144, 0.684, 3.42),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "nvidia/nemotron-3-ultra-550b-a55b:free",
|
||||
displayName: "Nemotron 3 Ultra (free)",
|
||||
@@ -354,6 +398,18 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
// Neither the OpenRouter page nor AgentHub's registry publishes a cache price for this
|
||||
// model, so cache_read repeats the input price (no discount assumed).
|
||||
modelId: "qwen/qwen3.6-35b-a3b",
|
||||
displayName: "Qwen 3.6 35B A3B",
|
||||
provider: "openrouter",
|
||||
contextWindow: 262144,
|
||||
pricing: usd(0.14, 0.14, 1),
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
// No official separate cache price published: cache_read uses the standard input price.
|
||||
modelId: "stepfun/step-3.7-flash",
|
||||
@@ -405,6 +461,16 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "z-ai/glm-5.1",
|
||||
displayName: "GLM-5.1",
|
||||
provider: "openrouter",
|
||||
contextWindow: 204800,
|
||||
pricing: usd(0.1794, 0.966, 3.036),
|
||||
supportsVision: false,
|
||||
clientType: "openai",
|
||||
baseUrl: OPENROUTER_BASE_URL,
|
||||
},
|
||||
// -- Fireworks AI (gateway, standard serverless USD pricing: cached input / uncached
|
||||
// input / output from each model's page; API ids use the accounts/fireworks/models/<slug>
|
||||
// form) --
|
||||
@@ -499,6 +565,38 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
clientType: "openai",
|
||||
baseUrl: SILICONFLOW_BASE_URL,
|
||||
},
|
||||
// The three Pro/ and Qwen/ entries below carry no pricing: AgentHub's registry publishes
|
||||
// none for them, and SiliconFlow's price list sits behind an authenticated API (the public
|
||||
// /v1/models endpoint returns 401 and the console page is client-rendered). Rather than
|
||||
// invent a rate, the entries ship unpriced — the same state as qwen3.8-max-preview, so their
|
||||
// cost reads as 0 until a published price can be filled in.
|
||||
{
|
||||
modelId: "Pro/moonshotai/Kimi-K2.6",
|
||||
displayName: "Kimi K2.6",
|
||||
provider: "siliconflow",
|
||||
contextWindow: 262144,
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: SILICONFLOW_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "Pro/zai-org/GLM-5.1",
|
||||
displayName: "GLM-5.1",
|
||||
provider: "siliconflow",
|
||||
contextWindow: 200000,
|
||||
supportsVision: false,
|
||||
clientType: "openai",
|
||||
baseUrl: SILICONFLOW_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "Qwen/Qwen3.6-35B-A3B",
|
||||
displayName: "Qwen 3.6 35B A3B",
|
||||
provider: "siliconflow",
|
||||
contextWindow: 262144,
|
||||
supportsVision: true,
|
||||
clientType: "openai",
|
||||
baseUrl: SILICONFLOW_BASE_URL,
|
||||
},
|
||||
{
|
||||
modelId: "zai-org/GLM-5.2",
|
||||
displayName: "GLM-5.2",
|
||||
@@ -608,6 +706,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
baseUrl: QWEN_PAYG_BASE_URL,
|
||||
},
|
||||
// -- Google Gemini (official USD pricing) --
|
||||
{
|
||||
modelId: "gemini-3.6-flash",
|
||||
displayName: "Gemini 3.6 Flash",
|
||||
provider: "google",
|
||||
contextWindow: 1048576,
|
||||
pricing: usd(0.15, 1.5, 7.5),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "gemini-3.5-flash",
|
||||
displayName: "Gemini 3.5 Flash",
|
||||
@@ -616,6 +722,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
pricing: usd(0.15, 1.5, 9),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "gemini-3.5-flash-lite",
|
||||
displayName: "Gemini 3.5 Flash-Lite",
|
||||
provider: "google",
|
||||
contextWindow: 1048576,
|
||||
pricing: usd(0.03, 0.3, 2.5),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "gemini-3.1-flash-lite",
|
||||
displayName: "Gemini 3.1 Flash-Lite",
|
||||
@@ -642,6 +756,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
supportsVision: true,
|
||||
},
|
||||
// -- Anthropic (official USD pricing; cache write = 1.25 x input) --
|
||||
{
|
||||
modelId: "claude-fable-5",
|
||||
displayName: "Claude Fable 5",
|
||||
provider: "anthropic",
|
||||
contextWindow: 1000000,
|
||||
pricing: usd(1, 12.5, 50),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "claude-opus-4-8",
|
||||
displayName: "Claude Opus 4.8",
|
||||
@@ -658,6 +780,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
pricing: usd(0.5, 6.25, 25),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "claude-sonnet-5",
|
||||
displayName: "Claude Sonnet 5",
|
||||
provider: "anthropic",
|
||||
contextWindow: 1000000,
|
||||
pricing: usd(0.2, 2.5, 10),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "claude-sonnet-4-6",
|
||||
displayName: "Claude Sonnet 4.6",
|
||||
@@ -743,6 +873,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
|
||||
supportsVision: false,
|
||||
},
|
||||
// -- Moonshot (Kimi) (official CNY pricing) --
|
||||
{
|
||||
modelId: "kimi-k3",
|
||||
displayName: "Kimi K3",
|
||||
provider: "moonshot",
|
||||
contextWindow: 1048576,
|
||||
pricing: cny(2, 20, 100),
|
||||
supportsVision: true,
|
||||
},
|
||||
{
|
||||
modelId: "kimi-k2.6",
|
||||
displayName: "Kimi K2.6",
|
||||
@@ -782,7 +920,7 @@ export interface ModelEnvInfo {
|
||||
|
||||
/**
|
||||
* Resolves the env var fallback for a model: mirrors AgentHub's
|
||||
* AutoLLMClient routing rules (verified against agenthub v0.3.3 autoClient.ts) - an explicit
|
||||
* AutoLLMClient routing rules (verified against agenthub v0.4.1 autoClient.ts) - an explicit
|
||||
* client_type takes priority, otherwise routes to a client by lowercase substring match on
|
||||
* model_id, returning the var pair that client reads; branch order matches AutoLLMClient.
|
||||
* Returns undefined on no match (AgentHub will reject that id: it needs an explicit
|
||||
@@ -803,6 +941,8 @@ export function resolveModelEnv(modelId: string, clientType?: string): ModelEnvI
|
||||
}
|
||||
if (t.includes("gpt-5.4") || t.includes("gpt-5.5")) return env("OPENAI");
|
||||
if (t.includes("glm-5")) return env("ZAI");
|
||||
// agenthub 0.4.1 routes kimi-k3 to its own client, which reads the same MOONSHOT_* pair.
|
||||
if (t.includes("kimi-k3")) return env("MOONSHOT");
|
||||
if (t.includes("kimi-k2.5") || t.includes("kimi-k2.6")) return env("MOONSHOT");
|
||||
if (t.includes("deepseek-v4")) return env("DEEPSEEK");
|
||||
if (t.includes("openai")) return env("OPENAI");
|
||||
|
||||
@@ -69,11 +69,19 @@ describe("model-catalog", () => {
|
||||
});
|
||||
|
||||
it("price buckets are positive (preview models without a list price omit pricing); context_window is a positive integer", () => {
|
||||
// Models with no obtainable published price. qwen3.8-max-preview: the plan runs a
|
||||
// quota-multiplier promotion instead of a per-token list price. The three SiliconFlow
|
||||
// entries: AgentHub's registry publishes no pricing for them and SiliconFlow's price list
|
||||
// is only reachable with an authenticated token, so no number can be sourced. All of them
|
||||
// carry no pricing and their costs read as 0, same as unpriced user models.
|
||||
const UNPRICED = new Set([
|
||||
"qwen-token-plan\0qwen3.8-max-preview",
|
||||
"siliconflow\0Pro/moonshotai/Kimi-K2.6",
|
||||
"siliconflow\0Pro/zai-org/GLM-5.1",
|
||||
"siliconflow\0Qwen/Qwen3.6-35B-A3B",
|
||||
]);
|
||||
for (const m of MODEL_CATALOG) {
|
||||
if (m.provider === "qwen-token-plan" && m.modelId === "qwen3.8-max-preview") {
|
||||
// Preview-only model: the plan runs a quota-multiplier promotion and publishes no
|
||||
// per-token list price, so the entry carries none and costs read as 0 (same as
|
||||
// unpriced user models).
|
||||
if (UNPRICED.has(`${m.provider}\0${m.modelId}`)) {
|
||||
expect(m.pricing, m.modelId).toBeUndefined();
|
||||
} else if (m.modelId.endsWith(":free")) {
|
||||
// Free-tier gateway model: a genuine $0 price (not "unknown"), so costs compute to 0.
|
||||
@@ -149,18 +157,23 @@ describe("model-catalog", () => {
|
||||
"anthropic/claude-sonnet-5",
|
||||
"deepseek/deepseek-v4-flash",
|
||||
"deepseek/deepseek-v4-pro",
|
||||
"google/gemini-3.6-flash",
|
||||
"google/gemini-3.5-flash",
|
||||
"google/gemini-3.5-flash-lite",
|
||||
"minimax/minimax-m3",
|
||||
"moonshotai/kimi-k3",
|
||||
"moonshotai/kimi-k2.6",
|
||||
"nvidia/nemotron-3-ultra-550b-a55b:free",
|
||||
"openai/gpt-5.6-sol",
|
||||
"openai/gpt-5.6-terra",
|
||||
"openai/gpt-5.5",
|
||||
"qwen/qwen3.6-35b-a3b",
|
||||
"stepfun/step-3.7-flash",
|
||||
"tencent/hy3",
|
||||
"x-ai/grok-4.5",
|
||||
"xiaomi/mimo-v2.5",
|
||||
"z-ai/glm-5.2",
|
||||
"z-ai/glm-5.1",
|
||||
]);
|
||||
for (const m of or) {
|
||||
expect(m.clientType).toBe("openai");
|
||||
@@ -179,11 +192,16 @@ describe("model-catalog", () => {
|
||||
expect(m.baseUrl).toBe("https://api.fireworks.ai/inference/v1");
|
||||
}
|
||||
const sf = MODEL_CATALOG.filter((m) => m.provider === "siliconflow");
|
||||
// Dictionary order is case-insensitive (as in qwen-pay-as-you-go, where ZHIPU/GLM-5.2
|
||||
// sorts last): Pro/ and Qwen/ fall between moonshotai/ and zai-org/.
|
||||
expect(sf.map((m) => m.modelId)).toEqual([
|
||||
"deepseek-ai/DeepSeek-V4-Flash",
|
||||
"deepseek-ai/DeepSeek-V4-Pro",
|
||||
"meituan-longcat/LongCat-2.0",
|
||||
"moonshotai/Kimi-K2.7-Code",
|
||||
"Pro/moonshotai/Kimi-K2.6",
|
||||
"Pro/zai-org/GLM-5.1",
|
||||
"Qwen/Qwen3.6-35B-A3B",
|
||||
"zai-org/GLM-5.2",
|
||||
]);
|
||||
for (const m of sf) {
|
||||
@@ -261,6 +279,25 @@ describe("model-catalog", () => {
|
||||
expect([mimo.cache_read, mimo.cache_write, mimo.output]).toEqual([0.0028, 0.14, 0.28]);
|
||||
const hy3 = MODEL_CATALOG.find((m) => m.modelId === "tencent/hy3")!.pricing!;
|
||||
expect([hy3.cache_read, hy3.cache_write, hy3.output]).toEqual([0.035, 0.14, 0.58]);
|
||||
// Gemini 3.6 Flash and 3.5 Flash Lite: upstream publishes a cache-hit price, so cache_read
|
||||
// stores the real discounted price (not the input price) — cache_read is its own billing
|
||||
// bucket in the cost center. cache_write repeats input (no per-token cache-write fee).
|
||||
const g36 = catalogEntryFor("openrouter", "google/gemini-3.6-flash")!;
|
||||
expect([g36.contextWindow, g36.supportsVision]).toEqual([1048576, true]);
|
||||
expect([g36.pricing!.cache_read, g36.pricing!.cache_write, g36.pricing!.output]).toEqual([
|
||||
0.15, 1.5, 7.5,
|
||||
]);
|
||||
const g35lite = catalogEntryFor("openrouter", "google/gemini-3.5-flash-lite")!;
|
||||
expect([g35lite.contextWindow, g35lite.supportsVision]).toEqual([1048576, true]);
|
||||
expect([
|
||||
g35lite.pricing!.cache_read,
|
||||
g35lite.pricing!.cache_write,
|
||||
g35lite.pricing!.output,
|
||||
]).toEqual([0.03, 0.3, 2.5]);
|
||||
// The gateway row for gemini-3.5-flash reports the same context window as the
|
||||
// direct-vendor row for that model (and as AgentHub's registry): 1048576, not 1000000.
|
||||
expect(catalogEntryFor("openrouter", "google/gemini-3.5-flash")!.contextWindow).toBe(1048576);
|
||||
expect(catalogEntryFor("google", "gemini-3.5-flash")!.contextWindow).toBe(1048576);
|
||||
|
||||
// In preset entries, exactly the gateway models (and only them) inline base_url (no credentials).
|
||||
const withBaseUrl = presetModelEntries().filter((e) => e.base_url !== undefined);
|
||||
@@ -269,12 +306,78 @@ describe("model-catalog", () => {
|
||||
);
|
||||
});
|
||||
|
||||
it("direct-vendor groups: auto-routed (no client_type / base_url), newest series first", () => {
|
||||
// These groups' ids are auto-routed by AgentHub, so they carry neither client_type nor a
|
||||
// preset base URL — the opposite of the gateway groups above.
|
||||
for (const id of ["google", "anthropic", "moonshot"]) {
|
||||
for (const m of MODEL_CATALOG.filter((e) => e.provider === id)) {
|
||||
expect(m.clientType, m.modelId).toBeUndefined();
|
||||
expect(m.baseUrl, m.modelId).toBeUndefined();
|
||||
}
|
||||
}
|
||||
// Dictionary order by tier with newer versions of a tier first (same rule the OpenRouter
|
||||
// block follows for the identical Claude line-up).
|
||||
expect(MODEL_CATALOG.filter((m) => m.provider === "google").map((m) => m.modelId)).toEqual([
|
||||
"gemini-3.6-flash",
|
||||
"gemini-3.5-flash",
|
||||
"gemini-3.5-flash-lite",
|
||||
"gemini-3.1-flash-lite",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3-flash-preview",
|
||||
]);
|
||||
expect(MODEL_CATALOG.filter((m) => m.provider === "anthropic").map((m) => m.modelId)).toEqual([
|
||||
"claude-fable-5",
|
||||
"claude-opus-4-8",
|
||||
"claude-opus-4-7",
|
||||
"claude-sonnet-5",
|
||||
"claude-sonnet-4-6",
|
||||
]);
|
||||
expect(MODEL_CATALOG.filter((m) => m.provider === "moonshot").map((m) => m.modelId)).toEqual([
|
||||
"kimi-k3",
|
||||
"kimi-k2.6",
|
||||
"kimi-k2.5",
|
||||
]);
|
||||
// Anthropic keeps its cache_write = 1.25 x input convention for the Claude 5 line too
|
||||
// (registry input 10 and 2 -> 12.5 and 2.5), unlike every other group where cache_write
|
||||
// repeats the input price.
|
||||
const fable = catalogEntryFor("anthropic", "claude-fable-5")!;
|
||||
expect([fable.pricing!.cache_read, fable.pricing!.cache_write, fable.pricing!.output]).toEqual([
|
||||
1, 12.5, 50,
|
||||
]);
|
||||
const sonnet5 = catalogEntryFor("anthropic", "claude-sonnet-5")!;
|
||||
expect([
|
||||
sonnet5.pricing!.cache_read,
|
||||
sonnet5.pricing!.cache_write,
|
||||
sonnet5.pricing!.output,
|
||||
]).toEqual([0.2, 2.5, 10]);
|
||||
// The same model resold by a gateway keeps one display name across groups.
|
||||
for (const [directProvider, directId, gatewayProvider, gatewayId] of [
|
||||
["anthropic", "claude-fable-5", "openrouter", "anthropic/claude-fable-5"],
|
||||
["anthropic", "claude-sonnet-5", "openrouter", "anthropic/claude-sonnet-5"],
|
||||
["google", "gemini-3.5-flash-lite", "openrouter", "google/gemini-3.5-flash-lite"],
|
||||
["moonshot", "kimi-k3", "openrouter", "moonshotai/kimi-k3"],
|
||||
["moonshot", "kimi-k2.6", "openrouter", "moonshotai/kimi-k2.6"],
|
||||
["moonshot", "kimi-k2.6", "siliconflow", "Pro/moonshotai/Kimi-K2.6"],
|
||||
["zhipu", "glm-5.1", "openrouter", "z-ai/glm-5.1"],
|
||||
["zhipu", "glm-5.1", "siliconflow", "Pro/zai-org/GLM-5.1"],
|
||||
] as const) {
|
||||
expect(
|
||||
catalogEntryFor(gatewayProvider, gatewayId)!.displayName,
|
||||
`${gatewayProvider}/${gatewayId}`,
|
||||
).toBe(catalogEntryFor(directProvider, directId)!.displayName);
|
||||
}
|
||||
});
|
||||
|
||||
it("DeepSeek and Kimi are initialized from official CNY prices (stored in USD; x7 recovers the official price)", () => {
|
||||
const cnyOf = (usdV: number) => Math.round(usdV * 7 * 1000) / 1000;
|
||||
const pro = MODEL_CATALOG.find((m) => m.modelId === "deepseek-v4-pro")!.pricing!;
|
||||
expect([cnyOf(pro.cache_read), cnyOf(pro.cache_write), cnyOf(pro.output)]).toEqual([
|
||||
0.025, 3, 6,
|
||||
]);
|
||||
const k3 = MODEL_CATALOG.find(
|
||||
(m) => m.provider === "moonshot" && m.modelId === "kimi-k3",
|
||||
)!.pricing!;
|
||||
expect([cnyOf(k3.cache_read), cnyOf(k3.cache_write), cnyOf(k3.output)]).toEqual([2, 20, 100]);
|
||||
const k26 = MODEL_CATALOG.find((m) => m.modelId === "kimi-k2.6")!.pricing!;
|
||||
expect([cnyOf(k26.cache_read), cnyOf(k26.cache_write), cnyOf(k26.output)]).toEqual([
|
||||
1.1, 6.5, 27,
|
||||
@@ -291,6 +394,14 @@ describe("resolveModelEnv (PRN-021: env fallback resolved by AgentHub routing ru
|
||||
expect(resolveModelEnv("gpt-5.5-pro")?.envKey).toBe("OPENAI_API_KEY");
|
||||
expect(resolveModelEnv("glm-5.2")?.envKey).toBe("ZAI_API_KEY");
|
||||
expect(resolveModelEnv("kimi-k2.6")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL");
|
||||
// agenthub 0.4.1 routes these to their own clients; both read the same env pair as the
|
||||
// family they belong to, so the id must still resolve (kimi-k3 matches no k2.x substring).
|
||||
expect(resolveModelEnv("kimi-k3")?.envKey).toBe("MOONSHOT_API_KEY");
|
||||
expect(resolveModelEnv("kimi-k3")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL");
|
||||
expect(resolveModelEnv("gemini-3.6-flash")?.envKey).toBe("GEMINI_API_KEY");
|
||||
expect(resolveModelEnv("gemini-3.5-flash-lite")?.envKey).toBe("GEMINI_API_KEY");
|
||||
expect(resolveModelEnv("claude-fable-5")?.envKey).toBe("ANTHROPIC_API_KEY");
|
||||
expect(resolveModelEnv("claude-sonnet-5")?.envKey).toBe("ANTHROPIC_API_KEY");
|
||||
});
|
||||
|
||||
it("explicit client_type beats id: the openai protocol always uses OPENAI_* (independent of grouping)", () => {
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
---
|
||||
name: agenthub-models
|
||||
description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis and embeddings with one client.
|
||||
description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis, embeddings and the supported-model registry with one client.
|
||||
short_description: Call model APIs with one AgentHub client.
|
||||
short_description_zh: 用一个 AgentHub 客户端调用模型 API。
|
||||
version: 7
|
||||
updated: 2026-07-21T00:00:00Z
|
||||
version: 8
|
||||
updated: 2026-07-22T00:00:00Z
|
||||
---
|
||||
|
||||
# AgentHub Model APIs
|
||||
@@ -15,7 +15,7 @@ updated: 2026-07-21T00:00:00Z
|
||||
npm install @prismshadow/agenthub
|
||||
```
|
||||
|
||||
The only entry point is `AutoLLMClient`:
|
||||
The main entry point is `AutoLLMClient`:
|
||||
|
||||
```ts
|
||||
import { AutoLLMClient } from "@prismshadow/agenthub";
|
||||
@@ -23,7 +23,7 @@ import { AutoLLMClient } from "@prismshadow/agenthub";
|
||||
const client = new AutoLLMClient({ model: "<model_id>", apiKey: "<key>", baseUrl: "<url>", clientType: "<type>" });
|
||||
```
|
||||
|
||||
`apiKey`, `baseUrl` and `clientType` are optional (see routing below).
|
||||
`apiKey`, `baseUrl` and `clientType` are optional (see routing below). The package also exports `listSupportedModels` (the model registry) and the error classes `AgentHubError`, `UnsupportedParameterError`, `EmptyResponseError` and `ToolCallArgumentParseError`.
|
||||
|
||||
## Before you start
|
||||
|
||||
@@ -49,16 +49,23 @@ Use exact model ids. If an id is not in the table below and the user has not giv
|
||||
|
||||
| Family | Official IDs | Gateway variants |
|
||||
| ---------------- | --------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Gemini 3.6 | `gemini-3.6-flash`, `gemini-3.5-flash-lite` | — |
|
||||
| Gemini 3 | `gemini-3.1-pro-preview`, `gemini-3.5-flash`, `gemini-3.1-flash-lite` | — |
|
||||
| Gemini 3 image | `gemini-3.1-flash-image-preview`, `gemini-3-pro-image-preview` | — |
|
||||
| Gemini 3 image | `gemini-3.1-flash-image`, `gemini-3-pro-image-preview` | — |
|
||||
| Gemini 3 TTS | `gemini-3.1-flash-tts-preview` | — |
|
||||
| Gemini embedding | `gemini-embedding-2` | — |
|
||||
| Claude | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | — |
|
||||
| Claude 5 | `claude-fable-5`, `claude-sonnet-5` | OpenRouter `anthropic/claude-fable-5`, `anthropic/claude-sonnet-5` |
|
||||
| Claude 4 | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | OpenRouter `anthropic/claude-opus-4.8`, `anthropic/claude-opus-4.7` |
|
||||
| GPT | `gpt-5.4`, `gpt-5.4-mini`, `gpt-5.4-nano`, `gpt-5.5` | — |
|
||||
| OpenAI embedding | `text-embedding-3-small`, `text-embedding-3-large` | — |
|
||||
| Kimi K3 | `kimi-k3` | OpenRouter `moonshotai/kimi-k3` |
|
||||
| Kimi K2.6 | `kimi-k2.6` | OpenRouter `moonshotai/kimi-k2.6`; SiliconFlow `Pro/moonshotai/Kimi-K2.6` |
|
||||
| DeepSeek V4 | `deepseek-v4-pro`, `deepseek-v4-flash` | OpenRouter `deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`; SiliconFlow `deepseek-ai/DeepSeek-V4-Pro`, `deepseek-ai/DeepSeek-V4-Flash` |
|
||||
| GLM 5.2 | `glm-5.2` | OpenRouter `z-ai/glm-5.2`; SiliconFlow `zai-org/GLM-5.2` |
|
||||
| GLM 5.1 | `glm-5.1` | OpenRouter `z-ai/glm-5.1`; SiliconFlow `Pro/zai-org/GLM-5.1` |
|
||||
| Qwen 3.6 | — | OpenRouter `qwen/qwen3.6-35b-a3b`; SiliconFlow `Qwen/Qwen3.6-35B-A3B` |
|
||||
|
||||
The image endpoint dropped its preview suffix: `gemini-3.1-flash-image-preview` is deprecated, use `gemini-3.1-flash-image`.
|
||||
|
||||
Gateway model lists can be queried online:
|
||||
|
||||
@@ -67,9 +74,28 @@ curl https://openrouter.ai/api/v1/models
|
||||
curl --request GET --url https://api.siliconflow.cn/v1/models --header 'Authorization: Bearer <token>'
|
||||
```
|
||||
|
||||
## Supported-model registry
|
||||
|
||||
`listSupportedModels(currency?)` returns the models AgentHub itself knows how to route, so ids, endpoints, modalities, context windows and prices can be read from the package instead of being hardcoded:
|
||||
|
||||
```ts
|
||||
import { listSupportedModels } from "@prismshadow/agenthub";
|
||||
|
||||
for (const m of listSupportedModels()) {
|
||||
console.log(m.model, m.base_url, m.client, m.context_window, m.pricing?.prompt_tokens);
|
||||
}
|
||||
```
|
||||
|
||||
- Each `SupportedModel` is `{ model, base_url, client, input_modalities, output_modalities, context_window?, pricing? }`. The `(model, base_url, client)` triple maps straight onto the constructor: `new AutoLLMClient({ model, baseUrl: base_url, clientType: client })`.
|
||||
- Modalities are `"Text" | "Image" | "Video" | "Audio" | "Embed"`. Coverage includes the official vendor endpoints plus the OpenRouter and SiliconFlow gateways; `context_window` and `pricing` are omitted where the platform publishes no authoritative value (image and TTS models, for instance).
|
||||
- `pricing` is per million tokens, keyed by the same usage buckets as `usage_metadata`: `prompt_tokens` (non-cached input), `thoughts_tokens` / `response_tokens` (both the output price) and optional `cached_tokens` (cache-hit price). Values are stored in USD; pass `listSupportedModels("CNY")` to convert at 7 CNY/USD.
|
||||
|
||||
The registry is the curated current line-up, so prefer it when picking a model or estimating cost. It is narrower than the routing rules: older ids in the table above (`gpt-5.4`, `claude-opus-4-7`, `gemini-3.1-pro-preview`, `gemini-3.1-flash-lite`) still route fine but no longer appear in it.
|
||||
|
||||
## Routing and credentials
|
||||
|
||||
- Without `clientType`, the client auto-routes by model id substring: `gemini-3*`, `gemini-embedding`, `claude` 4-6/4-7/4-8, `gpt-5.4`/`gpt-5.5`, `glm-5`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint.
|
||||
- Without `clientType`, the client auto-routes by model id substring, in this order: `gemini-3.6` / `gemini-3.5-flash-lite`, then `gemini-3*` / `gemini-embedding`, `claude` 4-6/4-7/4-8/-5, `gpt-5.4`/`gpt-5.5`, `glm-5.2`, `glm-5`, `kimi-k3`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint.
|
||||
- Exception: a Gemini id served by an OpenAI-compatible gateway (e.g. OpenRouter's `google/gemini-3.6-flash`) still matches the Gemini substring and would auto-route to the Google protocol client. Pass `clientType: "openai"` explicitly for those.
|
||||
- For any other OpenAI chat-completion compatible model (e.g. Qwen series via OpenRouter or SiliconFlow), pass `clientType: "openai"` plus `baseUrl` (embeddings endpoints use a different client type — see Embeddings below).
|
||||
- API key: constructor parameter first, then the provider environment variable — `DEEPSEEK_API_KEY`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`, `MOONSHOT_API_KEY`. Base URLs read the same names with `_BASE_URL`.
|
||||
|
||||
@@ -87,9 +113,30 @@ for await (const event of client.streamingResponseStateful({
|
||||
```
|
||||
|
||||
- Each `event` is a `UniEvent`: `event_type` is `start` | `delta` | `stop`, and `content_items` carry the increments.
|
||||
- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`) and `tools`.
|
||||
- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`), `tool_choice`, `prompt_caching` and `tools`.
|
||||
- `streamingResponseStateful` keeps conversation history inside the client; manage it with `getHistory()` / `setHistory(history)` / `clearHistory()`. The stateless variant is `streamingResponse({ messages, config })`.
|
||||
|
||||
## Config parameters the model may reject
|
||||
|
||||
A config value the target client cannot honour throws `UnsupportedParameterError` (an `AgentHubError` carrying `client` and `parameter`) while building the request, before anything reaches the network:
|
||||
|
||||
```ts
|
||||
import { UnsupportedParameterError } from "@prismshadow/agenthub";
|
||||
|
||||
try {
|
||||
// ...
|
||||
} catch (err) {
|
||||
if (err instanceof UnsupportedParameterError) console.error(err.parameter, err.message);
|
||||
}
|
||||
```
|
||||
|
||||
- `thinking_level` never throws: every client maps each level onto the closest one the model supports. Kimi K3 reasons unconditionally, so `NONE` degrades to its lowest effort rather than disabling thinking; GLM-5.2 sends `reasoning_effort` alongside its `thinking` block and only `NONE` disables it.
|
||||
- `temperature` is rejected outright by Gemini 3.6 — that generation deprecated the sampling parameters, so the client refuses them instead of sending a value the API ignores. GPT-5.5, Claude 4.8/5, DeepSeek V4, Kimi K2.6 and Kimi K3 accept only the protocol default `1.0` and reject any other value. Gemini 3, Claude 4.6, GLM and the generic OpenAI client pass it through.
|
||||
- `tool_choice`: `"auto"` is safe everywhere. Claude accepts a single forced tool name; DeepSeek V4 and Kimi K2.6 allow `"auto"` / `"none"`; Kimi K3 adds `"required"` but refuses a specific tool; GLM only accepts `"auto"`.
|
||||
- `prompt_caching`: every client accepts `PromptCaching.ENABLE` and rejects the other values — caching is on by default and Kimi K3 caches context automatically.
|
||||
|
||||
Leave a parameter unset and the protocol default applies, which is the portable choice when a script must run against several families.
|
||||
|
||||
## Image generation
|
||||
|
||||
Use a Gemini image model (see Model IDs) and set `config.image_config` (optional `aspect_ratio`, and `image_size` of `"1K"` | `"2K"`):
|
||||
@@ -97,7 +144,7 @@ Use a Gemini image model (see Model IDs) and set `config.image_config` (optional
|
||||
```ts
|
||||
import fs from "node:fs";
|
||||
|
||||
const client = new AutoLLMClient({ model: "gemini-3.1-flash-image-preview" });
|
||||
const client = new AutoLLMClient({ model: "gemini-3.1-flash-image" });
|
||||
for await (const event of client.streamingResponseStateful({
|
||||
message: { role: "user", content_items: [{ type: "text", text: "A penguin on a glacier" }] },
|
||||
config: { image_config: { aspect_ratio: "16:9", image_size: "2K" } },
|
||||
|
||||
Generated
+7
-7
@@ -36,8 +36,8 @@ importers:
|
||||
packages/cli:
|
||||
dependencies:
|
||||
'@prismshadow/agenthub':
|
||||
specifier: ^0.4.0
|
||||
version: 0.4.0(ws@8.21.0)
|
||||
specifier: ^0.4.1
|
||||
version: 0.4.1(ws@8.21.0)
|
||||
'@prismshadow/penguin-core':
|
||||
specifier: workspace:*
|
||||
version: link:../core
|
||||
@@ -79,8 +79,8 @@ importers:
|
||||
packages/core:
|
||||
dependencies:
|
||||
'@prismshadow/agenthub':
|
||||
specifier: ^0.4.0
|
||||
version: 0.4.0(ws@8.21.0)
|
||||
specifier: ^0.4.1
|
||||
version: 0.4.1(ws@8.21.0)
|
||||
'@prismshadow/penguin-skills':
|
||||
specifier: workspace:*
|
||||
version: link:../skills
|
||||
@@ -1040,8 +1040,8 @@ packages:
|
||||
engines: {node: '>=18'}
|
||||
hasBin: true
|
||||
|
||||
'@prismshadow/agenthub@0.4.0':
|
||||
resolution: {integrity: sha512-P96fAoYEtAy+BQHsrdQ2Rw5oXbrR2IAXcN6S3SnpUUlwpVoH2kkdtbe4bTGTafys0jJeH7s1lCZDF7KRvdWOoQ==}
|
||||
'@prismshadow/agenthub@0.4.1':
|
||||
resolution: {integrity: sha512-YKidCwa4ZO0+BP5E+vg/p8HQAPrTvTDaFmoj5omUHi7OxXCXPMDxwodj4VPz/ntWMDsgjsePhzkZIfVmSbECkw==}
|
||||
|
||||
'@protobufjs/aspromise@1.1.2':
|
||||
resolution: {integrity: sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==}
|
||||
@@ -3678,7 +3678,7 @@ snapshots:
|
||||
dependencies:
|
||||
playwright: 1.61.1
|
||||
|
||||
'@prismshadow/agenthub@0.4.0(ws@8.21.0)':
|
||||
'@prismshadow/agenthub@0.4.1(ws@8.21.0)':
|
||||
dependencies:
|
||||
'@anthropic-ai/bedrock-sdk': 0.26.4
|
||||
'@anthropic-ai/sdk': 0.81.0
|
||||
|
||||
Reference in New Issue
Block a user