feat: AgentHub 0.4.1 and a model-catalog refresh across every provider group (#43)

Co-authored-by: Alice <alice@prismshadow.com>
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-07-23 02:02:54 +08:00
committed by GitHub
parent d01d0faf7a
commit 0f6828e57b
8 changed files with 334 additions and 38 deletions
+5 -6
View File
@@ -82,16 +82,15 @@ Four Skill groups ship in the box ([docs](https://penguin.ooo/docs/skills)); Age
| Model | Providers |
| ---------------- | -------------------------------------------------------------------------------- |
| DeepSeek V4 | DeepSeek, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan |
| Kimi K3 | OpenRouter, Qwen Pay-As-You-Go |
| Kimi K2.6 | Moonshot AI |
| Kimi K3 | Moonshot AI, OpenRouter, Qwen Pay-As-You-Go |
| GLM 5.2 | Z.AI, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go |
| Hunyuan 3 | OpenRouter |
| Qwen 3.8 Max | Qwen Token Plan (preview) |
| GPT 5.5 | OpenAI, OpenRouter |
| Gemini 3.5 Flash | Google Gemini, OpenRouter |
| Claude Opus 4.8 | Anthropic, OpenRouter |
| GPT 5.6 | OpenRouter |
| Gemini 3.6 Flash | Google Gemini, OpenRouter |
| Claude 5 | Anthropic, OpenRouter |
Any OpenAI-protocol endpoint is supported: pick a preset above, or point a custom endpoint at any of the 1000+ online and local models.
Each family's latest generation only — the app's **Models** page lists every built-in preset, and any OpenAI-protocol endpoint works too: pick a preset, or point a custom endpoint at any of the 1000+ online and local models.
## Requirements
+5 -6
View File
@@ -82,16 +82,15 @@ https://github.com/user-attachments/assets/aec49ae9-b743-467b-b247-37bedfeaa36e
| 模型 | 可用供应商 |
| ---------------- | -------------------------------------------------------------------------------- |
| DeepSeek V4 | DeepSeek, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan |
| Kimi K3 | OpenRouter, Qwen Pay-As-You-Go |
| Kimi K2.6 | Moonshot AI |
| Kimi K3 | Moonshot AI, OpenRouter, Qwen Pay-As-You-Go |
| GLM 5.2 | Z.AI, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go |
| Hunyuan 3 | OpenRouter |
| Qwen 3.8 Max | Qwen Token Plan(预览) |
| GPT 5.5 | OpenAI, OpenRouter |
| Gemini 3.5 Flash | Google Gemini, OpenRouter |
| Claude Opus 4.8 | Anthropic, OpenRouter |
| GPT 5.6 | OpenRouter |
| Gemini 3.6 Flash | Google Gemini, OpenRouter |
| Claude 5 | Anthropic, OpenRouter |
只要是 OpenAI 协议的端点都可以接入:从上表选择预置,或用自定义端点连接 1000+ 在线与本地模型。
上表每个系列只列最新一代,完整预置清单请在应用的**模型**页查看;只要是 OpenAI 协议的端点都可以接入:选择预置,或用自定义端点连接 1000+ 在线与本地模型。
## 系统需求
+1 -1
View File
@@ -22,7 +22,7 @@
"penguin": "tsx src/index.ts"
},
"dependencies": {
"@prismshadow/agenthub": "^0.4.0",
"@prismshadow/agenthub": "^0.4.1",
"@prismshadow/penguin-core": "workspace:*",
"@prismshadow/penguin-server": "workspace:*",
"@prismshadow/penguin-skills": "workspace:*",
+1 -1
View File
@@ -36,7 +36,7 @@
"build": "tsup"
},
"dependencies": {
"@prismshadow/agenthub": "^0.4.0",
"@prismshadow/agenthub": "^0.4.1",
"@prismshadow/penguin-skills": "workspace:*",
"smol-toml": "^1.3.0",
"yaml": "^2.5.0"
+143 -3
View File
@@ -222,7 +222,12 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
// 2026-07-20 list no cache pricing on their OpenRouter pages, so cache_read carries the
// standard input price; their :free tier stores a genuine $0 price (not "unknown"), so
// costs correctly compute to 0. GPT models are uniformly vision-capable (OpenAI
// product-line policy) even where the gateway page omits the modality. --
// product-line policy) even where the gateway page omits the modality.
// Two cache_read conventions coexist in this block: rows whose upstream publishes a
// cache-hit price store that real price (the google/gemini-3.6-flash and
// google/gemini-3.5-flash-lite entries below), while the 2026-07-20 rows still repeat the
// input price. Several of those older rows do have a published cache price upstream and
// should be re-read in one pass; until then treat their cache_read as an upper bound. --
{
modelId: "anthropic/claude-fable-5",
displayName: "Claude Fable 5",
@@ -283,16 +288,45 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// Unlike the older OpenRouter entries above, upstream **does** publish a cache-hit price
// for the Gemini rows (2026-07-22: $0.15/mtok here, agreed by the OpenRouter models API
// and AgentHub's own supported-model registry), so cache_read stores the real discounted
// price rather than repeating the input price: cache_read is billed as its own bucket in
// the cost center, and an input-priced cache_read overstates cache-heavy spend 10x.
// cache_write repeats the input price (no separate per-token cache-write fee), matching
// the direct-vendor Gemini rows below.
modelId: "google/gemini-3.6-flash",
displayName: "Gemini 3.6 Flash",
provider: "openrouter",
contextWindow: 1048576,
pricing: usd(0.15, 1.5, 7.5),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "google/gemini-3.5-flash",
displayName: "Gemini 3.5 Flash",
provider: "openrouter",
contextWindow: 1000000,
contextWindow: 1048576,
pricing: usd(1.5, 1.5, 9),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// Same published-cache-price convention as gemini-3.6-flash above (2026-07-22: $0.03/mtok
// cache hit, $0.30 input, $2.50 output).
modelId: "google/gemini-3.5-flash-lite",
displayName: "Gemini 3.5 Flash-Lite",
provider: "openrouter",
contextWindow: 1048576,
pricing: usd(0.03, 0.3, 2.5),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// No official separate cache price published: cache_read uses the standard input price (no discount assumed).
modelId: "minimax/minimax-m3",
@@ -314,6 +348,16 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "moonshotai/kimi-k2.6",
displayName: "Kimi K2.6",
provider: "openrouter",
contextWindow: 262144,
pricing: usd(0.144, 0.684, 3.42),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "nvidia/nemotron-3-ultra-550b-a55b:free",
displayName: "Nemotron 3 Ultra (free)",
@@ -354,6 +398,18 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// Neither the OpenRouter page nor AgentHub's registry publishes a cache price for this
// model, so cache_read repeats the input price (no discount assumed).
modelId: "qwen/qwen3.6-35b-a3b",
displayName: "Qwen 3.6 35B A3B",
provider: "openrouter",
contextWindow: 262144,
pricing: usd(0.14, 0.14, 1),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// No official separate cache price published: cache_read uses the standard input price.
modelId: "stepfun/step-3.7-flash",
@@ -405,6 +461,16 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "z-ai/glm-5.1",
displayName: "GLM-5.1",
provider: "openrouter",
contextWindow: 204800,
pricing: usd(0.1794, 0.966, 3.036),
supportsVision: false,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
// -- Fireworks AI (gateway, standard serverless USD pricing: cached input / uncached
// input / output from each model's page; API ids use the accounts/fireworks/models/<slug>
// form) --
@@ -499,6 +565,38 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
// The three Pro/ and Qwen/ entries below carry no pricing: AgentHub's registry publishes
// none for them, and SiliconFlow's price list sits behind an authenticated API (the public
// /v1/models endpoint returns 401 and the console page is client-rendered). Rather than
// invent a rate, the entries ship unpriced — the same state as qwen3.8-max-preview, so their
// cost reads as 0 until a published price can be filled in.
{
modelId: "Pro/moonshotai/Kimi-K2.6",
displayName: "Kimi K2.6",
provider: "siliconflow",
contextWindow: 262144,
supportsVision: true,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "Pro/zai-org/GLM-5.1",
displayName: "GLM-5.1",
provider: "siliconflow",
contextWindow: 200000,
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "Qwen/Qwen3.6-35B-A3B",
displayName: "Qwen 3.6 35B A3B",
provider: "siliconflow",
contextWindow: 262144,
supportsVision: true,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "zai-org/GLM-5.2",
displayName: "GLM-5.2",
@@ -608,6 +706,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
baseUrl: QWEN_PAYG_BASE_URL,
},
// -- Google Gemini (official USD pricing) --
{
modelId: "gemini-3.6-flash",
displayName: "Gemini 3.6 Flash",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.15, 1.5, 7.5),
supportsVision: true,
},
{
modelId: "gemini-3.5-flash",
displayName: "Gemini 3.5 Flash",
@@ -616,6 +722,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
pricing: usd(0.15, 1.5, 9),
supportsVision: true,
},
{
modelId: "gemini-3.5-flash-lite",
displayName: "Gemini 3.5 Flash-Lite",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.03, 0.3, 2.5),
supportsVision: true,
},
{
modelId: "gemini-3.1-flash-lite",
displayName: "Gemini 3.1 Flash-Lite",
@@ -642,6 +756,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
supportsVision: true,
},
// -- Anthropic (official USD pricing; cache write = 1.25 x input) --
{
modelId: "claude-fable-5",
displayName: "Claude Fable 5",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(1, 12.5, 50),
supportsVision: true,
},
{
modelId: "claude-opus-4-8",
displayName: "Claude Opus 4.8",
@@ -658,6 +780,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
pricing: usd(0.5, 6.25, 25),
supportsVision: true,
},
{
modelId: "claude-sonnet-5",
displayName: "Claude Sonnet 5",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.2, 2.5, 10),
supportsVision: true,
},
{
modelId: "claude-sonnet-4-6",
displayName: "Claude Sonnet 4.6",
@@ -743,6 +873,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [
supportsVision: false,
},
// -- Moonshot (Kimi) (official CNY pricing) --
{
modelId: "kimi-k3",
displayName: "Kimi K3",
provider: "moonshot",
contextWindow: 1048576,
pricing: cny(2, 20, 100),
supportsVision: true,
},
{
modelId: "kimi-k2.6",
displayName: "Kimi K2.6",
@@ -782,7 +920,7 @@ export interface ModelEnvInfo {
/**
* Resolves the env var fallback for a model: mirrors AgentHub's
* AutoLLMClient routing rules (verified against agenthub v0.3.3 autoClient.ts) - an explicit
* AutoLLMClient routing rules (verified against agenthub v0.4.1 autoClient.ts) - an explicit
* client_type takes priority, otherwise routes to a client by lowercase substring match on
* model_id, returning the var pair that client reads; branch order matches AutoLLMClient.
* Returns undefined on no match (AgentHub will reject that id: it needs an explicit
@@ -803,6 +941,8 @@ export function resolveModelEnv(modelId: string, clientType?: string): ModelEnvI
}
if (t.includes("gpt-5.4") || t.includes("gpt-5.5")) return env("OPENAI");
if (t.includes("glm-5")) return env("ZAI");
// agenthub 0.4.1 routes kimi-k3 to its own client, which reads the same MOONSHOT_* pair.
if (t.includes("kimi-k3")) return env("MOONSHOT");
if (t.includes("kimi-k2.5") || t.includes("kimi-k2.6")) return env("MOONSHOT");
if (t.includes("deepseek-v4")) return env("DEEPSEEK");
if (t.includes("openai")) return env("OPENAI");
+115 -4
View File
@@ -69,11 +69,19 @@ describe("model-catalog", () => {
});
it("price buckets are positive (preview models without a list price omit pricing); context_window is a positive integer", () => {
// Models with no obtainable published price. qwen3.8-max-preview: the plan runs a
// quota-multiplier promotion instead of a per-token list price. The three SiliconFlow
// entries: AgentHub's registry publishes no pricing for them and SiliconFlow's price list
// is only reachable with an authenticated token, so no number can be sourced. All of them
// carry no pricing and their costs read as 0, same as unpriced user models.
const UNPRICED = new Set([
"qwen-token-plan\0qwen3.8-max-preview",
"siliconflow\0Pro/moonshotai/Kimi-K2.6",
"siliconflow\0Pro/zai-org/GLM-5.1",
"siliconflow\0Qwen/Qwen3.6-35B-A3B",
]);
for (const m of MODEL_CATALOG) {
if (m.provider === "qwen-token-plan" && m.modelId === "qwen3.8-max-preview") {
// Preview-only model: the plan runs a quota-multiplier promotion and publishes no
// per-token list price, so the entry carries none and costs read as 0 (same as
// unpriced user models).
if (UNPRICED.has(`${m.provider}\0${m.modelId}`)) {
expect(m.pricing, m.modelId).toBeUndefined();
} else if (m.modelId.endsWith(":free")) {
// Free-tier gateway model: a genuine $0 price (not "unknown"), so costs compute to 0.
@@ -149,18 +157,23 @@ describe("model-catalog", () => {
"anthropic/claude-sonnet-5",
"deepseek/deepseek-v4-flash",
"deepseek/deepseek-v4-pro",
"google/gemini-3.6-flash",
"google/gemini-3.5-flash",
"google/gemini-3.5-flash-lite",
"minimax/minimax-m3",
"moonshotai/kimi-k3",
"moonshotai/kimi-k2.6",
"nvidia/nemotron-3-ultra-550b-a55b:free",
"openai/gpt-5.6-sol",
"openai/gpt-5.6-terra",
"openai/gpt-5.5",
"qwen/qwen3.6-35b-a3b",
"stepfun/step-3.7-flash",
"tencent/hy3",
"x-ai/grok-4.5",
"xiaomi/mimo-v2.5",
"z-ai/glm-5.2",
"z-ai/glm-5.1",
]);
for (const m of or) {
expect(m.clientType).toBe("openai");
@@ -179,11 +192,16 @@ describe("model-catalog", () => {
expect(m.baseUrl).toBe("https://api.fireworks.ai/inference/v1");
}
const sf = MODEL_CATALOG.filter((m) => m.provider === "siliconflow");
// Dictionary order is case-insensitive (as in qwen-pay-as-you-go, where ZHIPU/GLM-5.2
// sorts last): Pro/ and Qwen/ fall between moonshotai/ and zai-org/.
expect(sf.map((m) => m.modelId)).toEqual([
"deepseek-ai/DeepSeek-V4-Flash",
"deepseek-ai/DeepSeek-V4-Pro",
"meituan-longcat/LongCat-2.0",
"moonshotai/Kimi-K2.7-Code",
"Pro/moonshotai/Kimi-K2.6",
"Pro/zai-org/GLM-5.1",
"Qwen/Qwen3.6-35B-A3B",
"zai-org/GLM-5.2",
]);
for (const m of sf) {
@@ -261,6 +279,25 @@ describe("model-catalog", () => {
expect([mimo.cache_read, mimo.cache_write, mimo.output]).toEqual([0.0028, 0.14, 0.28]);
const hy3 = MODEL_CATALOG.find((m) => m.modelId === "tencent/hy3")!.pricing!;
expect([hy3.cache_read, hy3.cache_write, hy3.output]).toEqual([0.035, 0.14, 0.58]);
// Gemini 3.6 Flash and 3.5 Flash Lite: upstream publishes a cache-hit price, so cache_read
// stores the real discounted price (not the input price) — cache_read is its own billing
// bucket in the cost center. cache_write repeats input (no per-token cache-write fee).
const g36 = catalogEntryFor("openrouter", "google/gemini-3.6-flash")!;
expect([g36.contextWindow, g36.supportsVision]).toEqual([1048576, true]);
expect([g36.pricing!.cache_read, g36.pricing!.cache_write, g36.pricing!.output]).toEqual([
0.15, 1.5, 7.5,
]);
const g35lite = catalogEntryFor("openrouter", "google/gemini-3.5-flash-lite")!;
expect([g35lite.contextWindow, g35lite.supportsVision]).toEqual([1048576, true]);
expect([
g35lite.pricing!.cache_read,
g35lite.pricing!.cache_write,
g35lite.pricing!.output,
]).toEqual([0.03, 0.3, 2.5]);
// The gateway row for gemini-3.5-flash reports the same context window as the
// direct-vendor row for that model (and as AgentHub's registry): 1048576, not 1000000.
expect(catalogEntryFor("openrouter", "google/gemini-3.5-flash")!.contextWindow).toBe(1048576);
expect(catalogEntryFor("google", "gemini-3.5-flash")!.contextWindow).toBe(1048576);
// In preset entries, exactly the gateway models (and only them) inline base_url (no credentials).
const withBaseUrl = presetModelEntries().filter((e) => e.base_url !== undefined);
@@ -269,12 +306,78 @@ describe("model-catalog", () => {
);
});
it("direct-vendor groups: auto-routed (no client_type / base_url), newest series first", () => {
// These groups' ids are auto-routed by AgentHub, so they carry neither client_type nor a
// preset base URL — the opposite of the gateway groups above.
for (const id of ["google", "anthropic", "moonshot"]) {
for (const m of MODEL_CATALOG.filter((e) => e.provider === id)) {
expect(m.clientType, m.modelId).toBeUndefined();
expect(m.baseUrl, m.modelId).toBeUndefined();
}
}
// Dictionary order by tier with newer versions of a tier first (same rule the OpenRouter
// block follows for the identical Claude line-up).
expect(MODEL_CATALOG.filter((m) => m.provider === "google").map((m) => m.modelId)).toEqual([
"gemini-3.6-flash",
"gemini-3.5-flash",
"gemini-3.5-flash-lite",
"gemini-3.1-flash-lite",
"gemini-3.1-pro-preview",
"gemini-3-flash-preview",
]);
expect(MODEL_CATALOG.filter((m) => m.provider === "anthropic").map((m) => m.modelId)).toEqual([
"claude-fable-5",
"claude-opus-4-8",
"claude-opus-4-7",
"claude-sonnet-5",
"claude-sonnet-4-6",
]);
expect(MODEL_CATALOG.filter((m) => m.provider === "moonshot").map((m) => m.modelId)).toEqual([
"kimi-k3",
"kimi-k2.6",
"kimi-k2.5",
]);
// Anthropic keeps its cache_write = 1.25 x input convention for the Claude 5 line too
// (registry input 10 and 2 -> 12.5 and 2.5), unlike every other group where cache_write
// repeats the input price.
const fable = catalogEntryFor("anthropic", "claude-fable-5")!;
expect([fable.pricing!.cache_read, fable.pricing!.cache_write, fable.pricing!.output]).toEqual([
1, 12.5, 50,
]);
const sonnet5 = catalogEntryFor("anthropic", "claude-sonnet-5")!;
expect([
sonnet5.pricing!.cache_read,
sonnet5.pricing!.cache_write,
sonnet5.pricing!.output,
]).toEqual([0.2, 2.5, 10]);
// The same model resold by a gateway keeps one display name across groups.
for (const [directProvider, directId, gatewayProvider, gatewayId] of [
["anthropic", "claude-fable-5", "openrouter", "anthropic/claude-fable-5"],
["anthropic", "claude-sonnet-5", "openrouter", "anthropic/claude-sonnet-5"],
["google", "gemini-3.5-flash-lite", "openrouter", "google/gemini-3.5-flash-lite"],
["moonshot", "kimi-k3", "openrouter", "moonshotai/kimi-k3"],
["moonshot", "kimi-k2.6", "openrouter", "moonshotai/kimi-k2.6"],
["moonshot", "kimi-k2.6", "siliconflow", "Pro/moonshotai/Kimi-K2.6"],
["zhipu", "glm-5.1", "openrouter", "z-ai/glm-5.1"],
["zhipu", "glm-5.1", "siliconflow", "Pro/zai-org/GLM-5.1"],
] as const) {
expect(
catalogEntryFor(gatewayProvider, gatewayId)!.displayName,
`${gatewayProvider}/${gatewayId}`,
).toBe(catalogEntryFor(directProvider, directId)!.displayName);
}
});
it("DeepSeek and Kimi are initialized from official CNY prices (stored in USD; x7 recovers the official price)", () => {
const cnyOf = (usdV: number) => Math.round(usdV * 7 * 1000) / 1000;
const pro = MODEL_CATALOG.find((m) => m.modelId === "deepseek-v4-pro")!.pricing!;
expect([cnyOf(pro.cache_read), cnyOf(pro.cache_write), cnyOf(pro.output)]).toEqual([
0.025, 3, 6,
]);
const k3 = MODEL_CATALOG.find(
(m) => m.provider === "moonshot" && m.modelId === "kimi-k3",
)!.pricing!;
expect([cnyOf(k3.cache_read), cnyOf(k3.cache_write), cnyOf(k3.output)]).toEqual([2, 20, 100]);
const k26 = MODEL_CATALOG.find((m) => m.modelId === "kimi-k2.6")!.pricing!;
expect([cnyOf(k26.cache_read), cnyOf(k26.cache_write), cnyOf(k26.output)]).toEqual([
1.1, 6.5, 27,
@@ -291,6 +394,14 @@ describe("resolveModelEnv (PRN-021: env fallback resolved by AgentHub routing ru
expect(resolveModelEnv("gpt-5.5-pro")?.envKey).toBe("OPENAI_API_KEY");
expect(resolveModelEnv("glm-5.2")?.envKey).toBe("ZAI_API_KEY");
expect(resolveModelEnv("kimi-k2.6")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL");
// agenthub 0.4.1 routes these to their own clients; both read the same env pair as the
// family they belong to, so the id must still resolve (kimi-k3 matches no k2.x substring).
expect(resolveModelEnv("kimi-k3")?.envKey).toBe("MOONSHOT_API_KEY");
expect(resolveModelEnv("kimi-k3")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL");
expect(resolveModelEnv("gemini-3.6-flash")?.envKey).toBe("GEMINI_API_KEY");
expect(resolveModelEnv("gemini-3.5-flash-lite")?.envKey).toBe("GEMINI_API_KEY");
expect(resolveModelEnv("claude-fable-5")?.envKey).toBe("ANTHROPIC_API_KEY");
expect(resolveModelEnv("claude-sonnet-5")?.envKey).toBe("ANTHROPIC_API_KEY");
});
it("explicit client_type beats id: the openai protocol always uses OPENAI_* (independent of grouping)", () => {
+57 -10
View File
@@ -1,10 +1,10 @@
---
name: agenthub-models
description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis and embeddings with one client.
description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis, embeddings and the supported-model registry with one client.
short_description: Call model APIs with one AgentHub client.
short_description_zh: 用一个 AgentHub 客户端调用模型 API。
version: 7
updated: 2026-07-21T00:00:00Z
version: 8
updated: 2026-07-22T00:00:00Z
---
# AgentHub Model APIs
@@ -15,7 +15,7 @@ updated: 2026-07-21T00:00:00Z
npm install @prismshadow/agenthub
```
The only entry point is `AutoLLMClient`:
The main entry point is `AutoLLMClient`:
```ts
import { AutoLLMClient } from "@prismshadow/agenthub";
@@ -23,7 +23,7 @@ import { AutoLLMClient } from "@prismshadow/agenthub";
const client = new AutoLLMClient({ model: "<model_id>", apiKey: "<key>", baseUrl: "<url>", clientType: "<type>" });
```
`apiKey`, `baseUrl` and `clientType` are optional (see routing below).
`apiKey`, `baseUrl` and `clientType` are optional (see routing below). The package also exports `listSupportedModels` (the model registry) and the error classes `AgentHubError`, `UnsupportedParameterError`, `EmptyResponseError` and `ToolCallArgumentParseError`.
## Before you start
@@ -49,16 +49,23 @@ Use exact model ids. If an id is not in the table below and the user has not giv
| Family | Official IDs | Gateway variants |
| ---------------- | --------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- |
| Gemini 3.6 | `gemini-3.6-flash`, `gemini-3.5-flash-lite` | — |
| Gemini 3 | `gemini-3.1-pro-preview`, `gemini-3.5-flash`, `gemini-3.1-flash-lite` | — |
| Gemini 3 image | `gemini-3.1-flash-image-preview`, `gemini-3-pro-image-preview` | — |
| Gemini 3 image | `gemini-3.1-flash-image`, `gemini-3-pro-image-preview` | — |
| Gemini 3 TTS | `gemini-3.1-flash-tts-preview` | — |
| Gemini embedding | `gemini-embedding-2` | — |
| Claude | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | — |
| Claude 5 | `claude-fable-5`, `claude-sonnet-5` | OpenRouter `anthropic/claude-fable-5`, `anthropic/claude-sonnet-5` |
| Claude 4 | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | OpenRouter `anthropic/claude-opus-4.8`, `anthropic/claude-opus-4.7` |
| GPT | `gpt-5.4`, `gpt-5.4-mini`, `gpt-5.4-nano`, `gpt-5.5` | — |
| OpenAI embedding | `text-embedding-3-small`, `text-embedding-3-large` | — |
| Kimi K3 | `kimi-k3` | OpenRouter `moonshotai/kimi-k3` |
| Kimi K2.6 | `kimi-k2.6` | OpenRouter `moonshotai/kimi-k2.6`; SiliconFlow `Pro/moonshotai/Kimi-K2.6` |
| DeepSeek V4 | `deepseek-v4-pro`, `deepseek-v4-flash` | OpenRouter `deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`; SiliconFlow `deepseek-ai/DeepSeek-V4-Pro`, `deepseek-ai/DeepSeek-V4-Flash` |
| GLM 5.2 | `glm-5.2` | OpenRouter `z-ai/glm-5.2`; SiliconFlow `zai-org/GLM-5.2` |
| GLM 5.1 | `glm-5.1` | OpenRouter `z-ai/glm-5.1`; SiliconFlow `Pro/zai-org/GLM-5.1` |
| Qwen 3.6 | — | OpenRouter `qwen/qwen3.6-35b-a3b`; SiliconFlow `Qwen/Qwen3.6-35B-A3B` |
The image endpoint dropped its preview suffix: `gemini-3.1-flash-image-preview` is deprecated, use `gemini-3.1-flash-image`.
Gateway model lists can be queried online:
@@ -67,9 +74,28 @@ curl https://openrouter.ai/api/v1/models
curl --request GET --url https://api.siliconflow.cn/v1/models --header 'Authorization: Bearer <token>'
```
## Supported-model registry
`listSupportedModels(currency?)` returns the models AgentHub itself knows how to route, so ids, endpoints, modalities, context windows and prices can be read from the package instead of being hardcoded:
```ts
import { listSupportedModels } from "@prismshadow/agenthub";
for (const m of listSupportedModels()) {
console.log(m.model, m.base_url, m.client, m.context_window, m.pricing?.prompt_tokens);
}
```
- Each `SupportedModel` is `{ model, base_url, client, input_modalities, output_modalities, context_window?, pricing? }`. The `(model, base_url, client)` triple maps straight onto the constructor: `new AutoLLMClient({ model, baseUrl: base_url, clientType: client })`.
- Modalities are `"Text" | "Image" | "Video" | "Audio" | "Embed"`. Coverage includes the official vendor endpoints plus the OpenRouter and SiliconFlow gateways; `context_window` and `pricing` are omitted where the platform publishes no authoritative value (image and TTS models, for instance).
- `pricing` is per million tokens, keyed by the same usage buckets as `usage_metadata`: `prompt_tokens` (non-cached input), `thoughts_tokens` / `response_tokens` (both the output price) and optional `cached_tokens` (cache-hit price). Values are stored in USD; pass `listSupportedModels("CNY")` to convert at 7 CNY/USD.
The registry is the curated current line-up, so prefer it when picking a model or estimating cost. It is narrower than the routing rules: older ids in the table above (`gpt-5.4`, `claude-opus-4-7`, `gemini-3.1-pro-preview`, `gemini-3.1-flash-lite`) still route fine but no longer appear in it.
## Routing and credentials
- Without `clientType`, the client auto-routes by model id substring: `gemini-3*`, `gemini-embedding`, `claude` 4-6/4-7/4-8, `gpt-5.4`/`gpt-5.5`, `glm-5`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint.
- Without `clientType`, the client auto-routes by model id substring, in this order: `gemini-3.6` / `gemini-3.5-flash-lite`, then `gemini-3*` / `gemini-embedding`, `claude` 4-6/4-7/4-8/-5, `gpt-5.4`/`gpt-5.5`, `glm-5.2`, `glm-5`, `kimi-k3`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint.
- Exception: a Gemini id served by an OpenAI-compatible gateway (e.g. OpenRouter's `google/gemini-3.6-flash`) still matches the Gemini substring and would auto-route to the Google protocol client. Pass `clientType: "openai"` explicitly for those.
- For any other OpenAI chat-completion compatible model (e.g. Qwen series via OpenRouter or SiliconFlow), pass `clientType: "openai"` plus `baseUrl` (embeddings endpoints use a different client type — see Embeddings below).
- API key: constructor parameter first, then the provider environment variable — `DEEPSEEK_API_KEY`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`, `MOONSHOT_API_KEY`. Base URLs read the same names with `_BASE_URL`.
@@ -87,9 +113,30 @@ for await (const event of client.streamingResponseStateful({
```
- Each `event` is a `UniEvent`: `event_type` is `start` | `delta` | `stop`, and `content_items` carry the increments.
- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`) and `tools`.
- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`), `tool_choice`, `prompt_caching` and `tools`.
- `streamingResponseStateful` keeps conversation history inside the client; manage it with `getHistory()` / `setHistory(history)` / `clearHistory()`. The stateless variant is `streamingResponse({ messages, config })`.
## Config parameters the model may reject
A config value the target client cannot honour throws `UnsupportedParameterError` (an `AgentHubError` carrying `client` and `parameter`) while building the request, before anything reaches the network:
```ts
import { UnsupportedParameterError } from "@prismshadow/agenthub";
try {
// ...
} catch (err) {
if (err instanceof UnsupportedParameterError) console.error(err.parameter, err.message);
}
```
- `thinking_level` never throws: every client maps each level onto the closest one the model supports. Kimi K3 reasons unconditionally, so `NONE` degrades to its lowest effort rather than disabling thinking; GLM-5.2 sends `reasoning_effort` alongside its `thinking` block and only `NONE` disables it.
- `temperature` is rejected outright by Gemini 3.6 — that generation deprecated the sampling parameters, so the client refuses them instead of sending a value the API ignores. GPT-5.5, Claude 4.8/5, DeepSeek V4, Kimi K2.6 and Kimi K3 accept only the protocol default `1.0` and reject any other value. Gemini 3, Claude 4.6, GLM and the generic OpenAI client pass it through.
- `tool_choice`: `"auto"` is safe everywhere. Claude accepts a single forced tool name; DeepSeek V4 and Kimi K2.6 allow `"auto"` / `"none"`; Kimi K3 adds `"required"` but refuses a specific tool; GLM only accepts `"auto"`.
- `prompt_caching`: every client accepts `PromptCaching.ENABLE` and rejects the other values — caching is on by default and Kimi K3 caches context automatically.
Leave a parameter unset and the protocol default applies, which is the portable choice when a script must run against several families.
## Image generation
Use a Gemini image model (see Model IDs) and set `config.image_config` (optional `aspect_ratio`, and `image_size` of `"1K"` | `"2K"`):
@@ -97,7 +144,7 @@ Use a Gemini image model (see Model IDs) and set `config.image_config` (optional
```ts
import fs from "node:fs";
const client = new AutoLLMClient({ model: "gemini-3.1-flash-image-preview" });
const client = new AutoLLMClient({ model: "gemini-3.1-flash-image" });
for await (const event of client.streamingResponseStateful({
message: { role: "user", content_items: [{ type: "text", text: "A penguin on a glacier" }] },
config: { image_config: { aspect_ratio: "16:9", image_size: "2K" } },
+7 -7
View File
@@ -36,8 +36,8 @@ importers:
packages/cli:
dependencies:
'@prismshadow/agenthub':
specifier: ^0.4.0
version: 0.4.0(ws@8.21.0)
specifier: ^0.4.1
version: 0.4.1(ws@8.21.0)
'@prismshadow/penguin-core':
specifier: workspace:*
version: link:../core
@@ -79,8 +79,8 @@ importers:
packages/core:
dependencies:
'@prismshadow/agenthub':
specifier: ^0.4.0
version: 0.4.0(ws@8.21.0)
specifier: ^0.4.1
version: 0.4.1(ws@8.21.0)
'@prismshadow/penguin-skills':
specifier: workspace:*
version: link:../skills
@@ -1040,8 +1040,8 @@ packages:
engines: {node: '>=18'}
hasBin: true
'@prismshadow/agenthub@0.4.0':
resolution: {integrity: sha512-P96fAoYEtAy+BQHsrdQ2Rw5oXbrR2IAXcN6S3SnpUUlwpVoH2kkdtbe4bTGTafys0jJeH7s1lCZDF7KRvdWOoQ==}
'@prismshadow/agenthub@0.4.1':
resolution: {integrity: sha512-YKidCwa4ZO0+BP5E+vg/p8HQAPrTvTDaFmoj5omUHi7OxXCXPMDxwodj4VPz/ntWMDsgjsePhzkZIfVmSbECkw==}
'@protobufjs/aspromise@1.1.2':
resolution: {integrity: sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==}
@@ -3678,7 +3678,7 @@ snapshots:
dependencies:
playwright: 1.61.1
'@prismshadow/agenthub@0.4.0(ws@8.21.0)':
'@prismshadow/agenthub@0.4.1(ws@8.21.0)':
dependencies:
'@anthropic-ai/bedrock-sdk': 0.26.4
'@anthropic-ai/sdk': 0.81.0