diff --git a/README.md b/README.md index 312aa6e..d85cbde 100644 --- a/README.md +++ b/README.md @@ -82,16 +82,15 @@ Four Skill groups ship in the box ([docs](https://penguin.ooo/docs/skills)); Age | Model | Providers | | ---------------- | -------------------------------------------------------------------------------- | | DeepSeek V4 | DeepSeek, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan | -| Kimi K3 | OpenRouter, Qwen Pay-As-You-Go | -| Kimi K2.6 | Moonshot AI | +| Kimi K3 | Moonshot AI, OpenRouter, Qwen Pay-As-You-Go | | GLM 5.2 | Z.AI, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go | | Hunyuan 3 | OpenRouter | | Qwen 3.8 Max | Qwen Token Plan (preview) | -| GPT 5.5 | OpenAI, OpenRouter | -| Gemini 3.5 Flash | Google Gemini, OpenRouter | -| Claude Opus 4.8 | Anthropic, OpenRouter | +| GPT 5.6 | OpenRouter | +| Gemini 3.6 Flash | Google Gemini, OpenRouter | +| Claude 5 | Anthropic, OpenRouter | -Any OpenAI-protocol endpoint is supported: pick a preset above, or point a custom endpoint at any of the 1000+ online and local models. +Each family's latest generation only — the app's **Models** page lists every built-in preset, and any OpenAI-protocol endpoint works too: pick a preset, or point a custom endpoint at any of the 1000+ online and local models. ## Requirements diff --git a/README.zh.md b/README.zh.md index dd3fa9f..d2c174b 100644 --- a/README.zh.md +++ b/README.zh.md @@ -82,16 +82,15 @@ https://github.com/user-attachments/assets/aec49ae9-b743-467b-b247-37bedfeaa36e | 模型 | 可用供应商 | | ---------------- | -------------------------------------------------------------------------------- | | DeepSeek V4 | DeepSeek, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan | -| Kimi K3 | OpenRouter, Qwen Pay-As-You-Go | -| Kimi K2.6 | Moonshot AI | +| Kimi K3 | Moonshot AI, OpenRouter, Qwen Pay-As-You-Go | | GLM 5.2 | Z.AI, OpenRouter, Fireworks AI, SiliconFlow, Qwen Token Plan, Qwen Pay-As-You-Go | | Hunyuan 3 | OpenRouter | | Qwen 3.8 Max | Qwen Token Plan(预览) | -| GPT 5.5 | OpenAI, OpenRouter | -| Gemini 3.5 Flash | Google Gemini, OpenRouter | -| Claude Opus 4.8 | Anthropic, OpenRouter | +| GPT 5.6 | OpenRouter | +| Gemini 3.6 Flash | Google Gemini, OpenRouter | +| Claude 5 | Anthropic, OpenRouter | -只要是 OpenAI 协议的端点都可以接入:从上表选择预置,或用自定义端点连接 1000+ 在线与本地模型。 +上表每个系列只列最新一代,完整预置清单请在应用的**模型**页查看;只要是 OpenAI 协议的端点都可以接入:选择预置,或用自定义端点连接 1000+ 在线与本地模型。 ## 系统需求 diff --git a/packages/cli/package.json b/packages/cli/package.json index fa1107f..44f1a8b 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -22,7 +22,7 @@ "penguin": "tsx src/index.ts" }, "dependencies": { - "@prismshadow/agenthub": "^0.4.0", + "@prismshadow/agenthub": "^0.4.1", "@prismshadow/penguin-core": "workspace:*", "@prismshadow/penguin-server": "workspace:*", "@prismshadow/penguin-skills": "workspace:*", diff --git a/packages/core/package.json b/packages/core/package.json index 2b88b58..eede7cb 100644 --- a/packages/core/package.json +++ b/packages/core/package.json @@ -36,7 +36,7 @@ "build": "tsup" }, "dependencies": { - "@prismshadow/agenthub": "^0.4.0", + "@prismshadow/agenthub": "^0.4.1", "@prismshadow/penguin-skills": "workspace:*", "smol-toml": "^1.3.0", "yaml": "^2.5.0" diff --git a/packages/core/src/state/model-catalog.ts b/packages/core/src/state/model-catalog.ts index 518ec4c..7b2c630 100644 --- a/packages/core/src/state/model-catalog.ts +++ b/packages/core/src/state/model-catalog.ts @@ -222,7 +222,12 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ // 2026-07-20 list no cache pricing on their OpenRouter pages, so cache_read carries the // standard input price; their :free tier stores a genuine $0 price (not "unknown"), so // costs correctly compute to 0. GPT models are uniformly vision-capable (OpenAI - // product-line policy) even where the gateway page omits the modality. -- + // product-line policy) even where the gateway page omits the modality. + // Two cache_read conventions coexist in this block: rows whose upstream publishes a + // cache-hit price store that real price (the google/gemini-3.6-flash and + // google/gemini-3.5-flash-lite entries below), while the 2026-07-20 rows still repeat the + // input price. Several of those older rows do have a published cache price upstream and + // should be re-read in one pass; until then treat their cache_read as an upper bound. -- { modelId: "anthropic/claude-fable-5", displayName: "Claude Fable 5", @@ -283,16 +288,45 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ clientType: "openai", baseUrl: OPENROUTER_BASE_URL, }, + { + // Unlike the older OpenRouter entries above, upstream **does** publish a cache-hit price + // for the Gemini rows (2026-07-22: $0.15/mtok here, agreed by the OpenRouter models API + // and AgentHub's own supported-model registry), so cache_read stores the real discounted + // price rather than repeating the input price: cache_read is billed as its own bucket in + // the cost center, and an input-priced cache_read overstates cache-heavy spend 10x. + // cache_write repeats the input price (no separate per-token cache-write fee), matching + // the direct-vendor Gemini rows below. + modelId: "google/gemini-3.6-flash", + displayName: "Gemini 3.6 Flash", + provider: "openrouter", + contextWindow: 1048576, + pricing: usd(0.15, 1.5, 7.5), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, { modelId: "google/gemini-3.5-flash", displayName: "Gemini 3.5 Flash", provider: "openrouter", - contextWindow: 1000000, + contextWindow: 1048576, pricing: usd(1.5, 1.5, 9), supportsVision: true, clientType: "openai", baseUrl: OPENROUTER_BASE_URL, }, + { + // Same published-cache-price convention as gemini-3.6-flash above (2026-07-22: $0.03/mtok + // cache hit, $0.30 input, $2.50 output). + modelId: "google/gemini-3.5-flash-lite", + displayName: "Gemini 3.5 Flash-Lite", + provider: "openrouter", + contextWindow: 1048576, + pricing: usd(0.03, 0.3, 2.5), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, { // No official separate cache price published: cache_read uses the standard input price (no discount assumed). modelId: "minimax/minimax-m3", @@ -314,6 +348,16 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ clientType: "openai", baseUrl: OPENROUTER_BASE_URL, }, + { + modelId: "moonshotai/kimi-k2.6", + displayName: "Kimi K2.6", + provider: "openrouter", + contextWindow: 262144, + pricing: usd(0.144, 0.684, 3.42), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, { modelId: "nvidia/nemotron-3-ultra-550b-a55b:free", displayName: "Nemotron 3 Ultra (free)", @@ -354,6 +398,18 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ clientType: "openai", baseUrl: OPENROUTER_BASE_URL, }, + { + // Neither the OpenRouter page nor AgentHub's registry publishes a cache price for this + // model, so cache_read repeats the input price (no discount assumed). + modelId: "qwen/qwen3.6-35b-a3b", + displayName: "Qwen 3.6 35B A3B", + provider: "openrouter", + contextWindow: 262144, + pricing: usd(0.14, 0.14, 1), + supportsVision: true, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, { // No official separate cache price published: cache_read uses the standard input price. modelId: "stepfun/step-3.7-flash", @@ -405,6 +461,16 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ clientType: "openai", baseUrl: OPENROUTER_BASE_URL, }, + { + modelId: "z-ai/glm-5.1", + displayName: "GLM-5.1", + provider: "openrouter", + contextWindow: 204800, + pricing: usd(0.1794, 0.966, 3.036), + supportsVision: false, + clientType: "openai", + baseUrl: OPENROUTER_BASE_URL, + }, // -- Fireworks AI (gateway, standard serverless USD pricing: cached input / uncached // input / output from each model's page; API ids use the accounts/fireworks/models/ // form) -- @@ -499,6 +565,38 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ clientType: "openai", baseUrl: SILICONFLOW_BASE_URL, }, + // The three Pro/ and Qwen/ entries below carry no pricing: AgentHub's registry publishes + // none for them, and SiliconFlow's price list sits behind an authenticated API (the public + // /v1/models endpoint returns 401 and the console page is client-rendered). Rather than + // invent a rate, the entries ship unpriced — the same state as qwen3.8-max-preview, so their + // cost reads as 0 until a published price can be filled in. + { + modelId: "Pro/moonshotai/Kimi-K2.6", + displayName: "Kimi K2.6", + provider: "siliconflow", + contextWindow: 262144, + supportsVision: true, + clientType: "openai", + baseUrl: SILICONFLOW_BASE_URL, + }, + { + modelId: "Pro/zai-org/GLM-5.1", + displayName: "GLM-5.1", + provider: "siliconflow", + contextWindow: 200000, + supportsVision: false, + clientType: "openai", + baseUrl: SILICONFLOW_BASE_URL, + }, + { + modelId: "Qwen/Qwen3.6-35B-A3B", + displayName: "Qwen 3.6 35B A3B", + provider: "siliconflow", + contextWindow: 262144, + supportsVision: true, + clientType: "openai", + baseUrl: SILICONFLOW_BASE_URL, + }, { modelId: "zai-org/GLM-5.2", displayName: "GLM-5.2", @@ -608,6 +706,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ baseUrl: QWEN_PAYG_BASE_URL, }, // -- Google Gemini (official USD pricing) -- + { + modelId: "gemini-3.6-flash", + displayName: "Gemini 3.6 Flash", + provider: "google", + contextWindow: 1048576, + pricing: usd(0.15, 1.5, 7.5), + supportsVision: true, + }, { modelId: "gemini-3.5-flash", displayName: "Gemini 3.5 Flash", @@ -616,6 +722,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ pricing: usd(0.15, 1.5, 9), supportsVision: true, }, + { + modelId: "gemini-3.5-flash-lite", + displayName: "Gemini 3.5 Flash-Lite", + provider: "google", + contextWindow: 1048576, + pricing: usd(0.03, 0.3, 2.5), + supportsVision: true, + }, { modelId: "gemini-3.1-flash-lite", displayName: "Gemini 3.1 Flash-Lite", @@ -642,6 +756,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ supportsVision: true, }, // -- Anthropic (official USD pricing; cache write = 1.25 x input) -- + { + modelId: "claude-fable-5", + displayName: "Claude Fable 5", + provider: "anthropic", + contextWindow: 1000000, + pricing: usd(1, 12.5, 50), + supportsVision: true, + }, { modelId: "claude-opus-4-8", displayName: "Claude Opus 4.8", @@ -658,6 +780,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ pricing: usd(0.5, 6.25, 25), supportsVision: true, }, + { + modelId: "claude-sonnet-5", + displayName: "Claude Sonnet 5", + provider: "anthropic", + contextWindow: 1000000, + pricing: usd(0.2, 2.5, 10), + supportsVision: true, + }, { modelId: "claude-sonnet-4-6", displayName: "Claude Sonnet 4.6", @@ -743,6 +873,14 @@ export const MODEL_CATALOG: ModelCatalogEntry[] = [ supportsVision: false, }, // -- Moonshot (Kimi) (official CNY pricing) -- + { + modelId: "kimi-k3", + displayName: "Kimi K3", + provider: "moonshot", + contextWindow: 1048576, + pricing: cny(2, 20, 100), + supportsVision: true, + }, { modelId: "kimi-k2.6", displayName: "Kimi K2.6", @@ -782,7 +920,7 @@ export interface ModelEnvInfo { /** * Resolves the env var fallback for a model: mirrors AgentHub's - * AutoLLMClient routing rules (verified against agenthub v0.3.3 autoClient.ts) - an explicit + * AutoLLMClient routing rules (verified against agenthub v0.4.1 autoClient.ts) - an explicit * client_type takes priority, otherwise routes to a client by lowercase substring match on * model_id, returning the var pair that client reads; branch order matches AutoLLMClient. * Returns undefined on no match (AgentHub will reject that id: it needs an explicit @@ -803,6 +941,8 @@ export function resolveModelEnv(modelId: string, clientType?: string): ModelEnvI } if (t.includes("gpt-5.4") || t.includes("gpt-5.5")) return env("OPENAI"); if (t.includes("glm-5")) return env("ZAI"); + // agenthub 0.4.1 routes kimi-k3 to its own client, which reads the same MOONSHOT_* pair. + if (t.includes("kimi-k3")) return env("MOONSHOT"); if (t.includes("kimi-k2.5") || t.includes("kimi-k2.6")) return env("MOONSHOT"); if (t.includes("deepseek-v4")) return env("DEEPSEEK"); if (t.includes("openai")) return env("OPENAI"); diff --git a/packages/core/test/model-catalog.test.ts b/packages/core/test/model-catalog.test.ts index eeafe6a..e5dfaa2 100644 --- a/packages/core/test/model-catalog.test.ts +++ b/packages/core/test/model-catalog.test.ts @@ -69,11 +69,19 @@ describe("model-catalog", () => { }); it("price buckets are positive (preview models without a list price omit pricing); context_window is a positive integer", () => { + // Models with no obtainable published price. qwen3.8-max-preview: the plan runs a + // quota-multiplier promotion instead of a per-token list price. The three SiliconFlow + // entries: AgentHub's registry publishes no pricing for them and SiliconFlow's price list + // is only reachable with an authenticated token, so no number can be sourced. All of them + // carry no pricing and their costs read as 0, same as unpriced user models. + const UNPRICED = new Set([ + "qwen-token-plan\0qwen3.8-max-preview", + "siliconflow\0Pro/moonshotai/Kimi-K2.6", + "siliconflow\0Pro/zai-org/GLM-5.1", + "siliconflow\0Qwen/Qwen3.6-35B-A3B", + ]); for (const m of MODEL_CATALOG) { - if (m.provider === "qwen-token-plan" && m.modelId === "qwen3.8-max-preview") { - // Preview-only model: the plan runs a quota-multiplier promotion and publishes no - // per-token list price, so the entry carries none and costs read as 0 (same as - // unpriced user models). + if (UNPRICED.has(`${m.provider}\0${m.modelId}`)) { expect(m.pricing, m.modelId).toBeUndefined(); } else if (m.modelId.endsWith(":free")) { // Free-tier gateway model: a genuine $0 price (not "unknown"), so costs compute to 0. @@ -149,18 +157,23 @@ describe("model-catalog", () => { "anthropic/claude-sonnet-5", "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-pro", + "google/gemini-3.6-flash", "google/gemini-3.5-flash", + "google/gemini-3.5-flash-lite", "minimax/minimax-m3", "moonshotai/kimi-k3", + "moonshotai/kimi-k2.6", "nvidia/nemotron-3-ultra-550b-a55b:free", "openai/gpt-5.6-sol", "openai/gpt-5.6-terra", "openai/gpt-5.5", + "qwen/qwen3.6-35b-a3b", "stepfun/step-3.7-flash", "tencent/hy3", "x-ai/grok-4.5", "xiaomi/mimo-v2.5", "z-ai/glm-5.2", + "z-ai/glm-5.1", ]); for (const m of or) { expect(m.clientType).toBe("openai"); @@ -179,11 +192,16 @@ describe("model-catalog", () => { expect(m.baseUrl).toBe("https://api.fireworks.ai/inference/v1"); } const sf = MODEL_CATALOG.filter((m) => m.provider === "siliconflow"); + // Dictionary order is case-insensitive (as in qwen-pay-as-you-go, where ZHIPU/GLM-5.2 + // sorts last): Pro/ and Qwen/ fall between moonshotai/ and zai-org/. expect(sf.map((m) => m.modelId)).toEqual([ "deepseek-ai/DeepSeek-V4-Flash", "deepseek-ai/DeepSeek-V4-Pro", "meituan-longcat/LongCat-2.0", "moonshotai/Kimi-K2.7-Code", + "Pro/moonshotai/Kimi-K2.6", + "Pro/zai-org/GLM-5.1", + "Qwen/Qwen3.6-35B-A3B", "zai-org/GLM-5.2", ]); for (const m of sf) { @@ -261,6 +279,25 @@ describe("model-catalog", () => { expect([mimo.cache_read, mimo.cache_write, mimo.output]).toEqual([0.0028, 0.14, 0.28]); const hy3 = MODEL_CATALOG.find((m) => m.modelId === "tencent/hy3")!.pricing!; expect([hy3.cache_read, hy3.cache_write, hy3.output]).toEqual([0.035, 0.14, 0.58]); + // Gemini 3.6 Flash and 3.5 Flash Lite: upstream publishes a cache-hit price, so cache_read + // stores the real discounted price (not the input price) — cache_read is its own billing + // bucket in the cost center. cache_write repeats input (no per-token cache-write fee). + const g36 = catalogEntryFor("openrouter", "google/gemini-3.6-flash")!; + expect([g36.contextWindow, g36.supportsVision]).toEqual([1048576, true]); + expect([g36.pricing!.cache_read, g36.pricing!.cache_write, g36.pricing!.output]).toEqual([ + 0.15, 1.5, 7.5, + ]); + const g35lite = catalogEntryFor("openrouter", "google/gemini-3.5-flash-lite")!; + expect([g35lite.contextWindow, g35lite.supportsVision]).toEqual([1048576, true]); + expect([ + g35lite.pricing!.cache_read, + g35lite.pricing!.cache_write, + g35lite.pricing!.output, + ]).toEqual([0.03, 0.3, 2.5]); + // The gateway row for gemini-3.5-flash reports the same context window as the + // direct-vendor row for that model (and as AgentHub's registry): 1048576, not 1000000. + expect(catalogEntryFor("openrouter", "google/gemini-3.5-flash")!.contextWindow).toBe(1048576); + expect(catalogEntryFor("google", "gemini-3.5-flash")!.contextWindow).toBe(1048576); // In preset entries, exactly the gateway models (and only them) inline base_url (no credentials). const withBaseUrl = presetModelEntries().filter((e) => e.base_url !== undefined); @@ -269,12 +306,78 @@ describe("model-catalog", () => { ); }); + it("direct-vendor groups: auto-routed (no client_type / base_url), newest series first", () => { + // These groups' ids are auto-routed by AgentHub, so they carry neither client_type nor a + // preset base URL — the opposite of the gateway groups above. + for (const id of ["google", "anthropic", "moonshot"]) { + for (const m of MODEL_CATALOG.filter((e) => e.provider === id)) { + expect(m.clientType, m.modelId).toBeUndefined(); + expect(m.baseUrl, m.modelId).toBeUndefined(); + } + } + // Dictionary order by tier with newer versions of a tier first (same rule the OpenRouter + // block follows for the identical Claude line-up). + expect(MODEL_CATALOG.filter((m) => m.provider === "google").map((m) => m.modelId)).toEqual([ + "gemini-3.6-flash", + "gemini-3.5-flash", + "gemini-3.5-flash-lite", + "gemini-3.1-flash-lite", + "gemini-3.1-pro-preview", + "gemini-3-flash-preview", + ]); + expect(MODEL_CATALOG.filter((m) => m.provider === "anthropic").map((m) => m.modelId)).toEqual([ + "claude-fable-5", + "claude-opus-4-8", + "claude-opus-4-7", + "claude-sonnet-5", + "claude-sonnet-4-6", + ]); + expect(MODEL_CATALOG.filter((m) => m.provider === "moonshot").map((m) => m.modelId)).toEqual([ + "kimi-k3", + "kimi-k2.6", + "kimi-k2.5", + ]); + // Anthropic keeps its cache_write = 1.25 x input convention for the Claude 5 line too + // (registry input 10 and 2 -> 12.5 and 2.5), unlike every other group where cache_write + // repeats the input price. + const fable = catalogEntryFor("anthropic", "claude-fable-5")!; + expect([fable.pricing!.cache_read, fable.pricing!.cache_write, fable.pricing!.output]).toEqual([ + 1, 12.5, 50, + ]); + const sonnet5 = catalogEntryFor("anthropic", "claude-sonnet-5")!; + expect([ + sonnet5.pricing!.cache_read, + sonnet5.pricing!.cache_write, + sonnet5.pricing!.output, + ]).toEqual([0.2, 2.5, 10]); + // The same model resold by a gateway keeps one display name across groups. + for (const [directProvider, directId, gatewayProvider, gatewayId] of [ + ["anthropic", "claude-fable-5", "openrouter", "anthropic/claude-fable-5"], + ["anthropic", "claude-sonnet-5", "openrouter", "anthropic/claude-sonnet-5"], + ["google", "gemini-3.5-flash-lite", "openrouter", "google/gemini-3.5-flash-lite"], + ["moonshot", "kimi-k3", "openrouter", "moonshotai/kimi-k3"], + ["moonshot", "kimi-k2.6", "openrouter", "moonshotai/kimi-k2.6"], + ["moonshot", "kimi-k2.6", "siliconflow", "Pro/moonshotai/Kimi-K2.6"], + ["zhipu", "glm-5.1", "openrouter", "z-ai/glm-5.1"], + ["zhipu", "glm-5.1", "siliconflow", "Pro/zai-org/GLM-5.1"], + ] as const) { + expect( + catalogEntryFor(gatewayProvider, gatewayId)!.displayName, + `${gatewayProvider}/${gatewayId}`, + ).toBe(catalogEntryFor(directProvider, directId)!.displayName); + } + }); + it("DeepSeek and Kimi are initialized from official CNY prices (stored in USD; x7 recovers the official price)", () => { const cnyOf = (usdV: number) => Math.round(usdV * 7 * 1000) / 1000; const pro = MODEL_CATALOG.find((m) => m.modelId === "deepseek-v4-pro")!.pricing!; expect([cnyOf(pro.cache_read), cnyOf(pro.cache_write), cnyOf(pro.output)]).toEqual([ 0.025, 3, 6, ]); + const k3 = MODEL_CATALOG.find( + (m) => m.provider === "moonshot" && m.modelId === "kimi-k3", + )!.pricing!; + expect([cnyOf(k3.cache_read), cnyOf(k3.cache_write), cnyOf(k3.output)]).toEqual([2, 20, 100]); const k26 = MODEL_CATALOG.find((m) => m.modelId === "kimi-k2.6")!.pricing!; expect([cnyOf(k26.cache_read), cnyOf(k26.cache_write), cnyOf(k26.output)]).toEqual([ 1.1, 6.5, 27, @@ -291,6 +394,14 @@ describe("resolveModelEnv (PRN-021: env fallback resolved by AgentHub routing ru expect(resolveModelEnv("gpt-5.5-pro")?.envKey).toBe("OPENAI_API_KEY"); expect(resolveModelEnv("glm-5.2")?.envKey).toBe("ZAI_API_KEY"); expect(resolveModelEnv("kimi-k2.6")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL"); + // agenthub 0.4.1 routes these to their own clients; both read the same env pair as the + // family they belong to, so the id must still resolve (kimi-k3 matches no k2.x substring). + expect(resolveModelEnv("kimi-k3")?.envKey).toBe("MOONSHOT_API_KEY"); + expect(resolveModelEnv("kimi-k3")?.envBaseUrlKey).toBe("MOONSHOT_BASE_URL"); + expect(resolveModelEnv("gemini-3.6-flash")?.envKey).toBe("GEMINI_API_KEY"); + expect(resolveModelEnv("gemini-3.5-flash-lite")?.envKey).toBe("GEMINI_API_KEY"); + expect(resolveModelEnv("claude-fable-5")?.envKey).toBe("ANTHROPIC_API_KEY"); + expect(resolveModelEnv("claude-sonnet-5")?.envKey).toBe("ANTHROPIC_API_KEY"); }); it("explicit client_type beats id: the openai protocol always uses OPENAI_* (independent of grouping)", () => { diff --git a/packages/skills/skills/agenthub-models/SKILL.md b/packages/skills/skills/agenthub-models/SKILL.md index 6890b6b..558b0ff 100644 --- a/packages/skills/skills/agenthub-models/SKILL.md +++ b/packages/skills/skills/agenthub-models/SKILL.md @@ -1,10 +1,10 @@ --- name: agenthub-models -description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis and embeddings with one client. +description: Call model APIs through @prismshadow/agenthub — streaming text generation, image generation, speech synthesis, embeddings and the supported-model registry with one client. short_description: Call model APIs with one AgentHub client. short_description_zh: 用一个 AgentHub 客户端调用模型 API。 -version: 7 -updated: 2026-07-21T00:00:00Z +version: 8 +updated: 2026-07-22T00:00:00Z --- # AgentHub Model APIs @@ -15,7 +15,7 @@ updated: 2026-07-21T00:00:00Z npm install @prismshadow/agenthub ``` -The only entry point is `AutoLLMClient`: +The main entry point is `AutoLLMClient`: ```ts import { AutoLLMClient } from "@prismshadow/agenthub"; @@ -23,7 +23,7 @@ import { AutoLLMClient } from "@prismshadow/agenthub"; const client = new AutoLLMClient({ model: "", apiKey: "", baseUrl: "", clientType: "" }); ``` -`apiKey`, `baseUrl` and `clientType` are optional (see routing below). +`apiKey`, `baseUrl` and `clientType` are optional (see routing below). The package also exports `listSupportedModels` (the model registry) and the error classes `AgentHubError`, `UnsupportedParameterError`, `EmptyResponseError` and `ToolCallArgumentParseError`. ## Before you start @@ -49,16 +49,23 @@ Use exact model ids. If an id is not in the table below and the user has not giv | Family | Official IDs | Gateway variants | | ---------------- | --------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- | +| Gemini 3.6 | `gemini-3.6-flash`, `gemini-3.5-flash-lite` | — | | Gemini 3 | `gemini-3.1-pro-preview`, `gemini-3.5-flash`, `gemini-3.1-flash-lite` | — | -| Gemini 3 image | `gemini-3.1-flash-image-preview`, `gemini-3-pro-image-preview` | — | +| Gemini 3 image | `gemini-3.1-flash-image`, `gemini-3-pro-image-preview` | — | | Gemini 3 TTS | `gemini-3.1-flash-tts-preview` | — | | Gemini embedding | `gemini-embedding-2` | — | -| Claude | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | — | +| Claude 5 | `claude-fable-5`, `claude-sonnet-5` | OpenRouter `anthropic/claude-fable-5`, `anthropic/claude-sonnet-5` | +| Claude 4 | `claude-sonnet-4-6`, `claude-opus-4-7`, `claude-opus-4-8` | OpenRouter `anthropic/claude-opus-4.8`, `anthropic/claude-opus-4.7` | | GPT | `gpt-5.4`, `gpt-5.4-mini`, `gpt-5.4-nano`, `gpt-5.5` | — | | OpenAI embedding | `text-embedding-3-small`, `text-embedding-3-large` | — | +| Kimi K3 | `kimi-k3` | OpenRouter `moonshotai/kimi-k3` | | Kimi K2.6 | `kimi-k2.6` | OpenRouter `moonshotai/kimi-k2.6`; SiliconFlow `Pro/moonshotai/Kimi-K2.6` | | DeepSeek V4 | `deepseek-v4-pro`, `deepseek-v4-flash` | OpenRouter `deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`; SiliconFlow `deepseek-ai/DeepSeek-V4-Pro`, `deepseek-ai/DeepSeek-V4-Flash` | +| GLM 5.2 | `glm-5.2` | OpenRouter `z-ai/glm-5.2`; SiliconFlow `zai-org/GLM-5.2` | | GLM 5.1 | `glm-5.1` | OpenRouter `z-ai/glm-5.1`; SiliconFlow `Pro/zai-org/GLM-5.1` | +| Qwen 3.6 | — | OpenRouter `qwen/qwen3.6-35b-a3b`; SiliconFlow `Qwen/Qwen3.6-35B-A3B` | + +The image endpoint dropped its preview suffix: `gemini-3.1-flash-image-preview` is deprecated, use `gemini-3.1-flash-image`. Gateway model lists can be queried online: @@ -67,9 +74,28 @@ curl https://openrouter.ai/api/v1/models curl --request GET --url https://api.siliconflow.cn/v1/models --header 'Authorization: Bearer ' ``` +## Supported-model registry + +`listSupportedModels(currency?)` returns the models AgentHub itself knows how to route, so ids, endpoints, modalities, context windows and prices can be read from the package instead of being hardcoded: + +```ts +import { listSupportedModels } from "@prismshadow/agenthub"; + +for (const m of listSupportedModels()) { + console.log(m.model, m.base_url, m.client, m.context_window, m.pricing?.prompt_tokens); +} +``` + +- Each `SupportedModel` is `{ model, base_url, client, input_modalities, output_modalities, context_window?, pricing? }`. The `(model, base_url, client)` triple maps straight onto the constructor: `new AutoLLMClient({ model, baseUrl: base_url, clientType: client })`. +- Modalities are `"Text" | "Image" | "Video" | "Audio" | "Embed"`. Coverage includes the official vendor endpoints plus the OpenRouter and SiliconFlow gateways; `context_window` and `pricing` are omitted where the platform publishes no authoritative value (image and TTS models, for instance). +- `pricing` is per million tokens, keyed by the same usage buckets as `usage_metadata`: `prompt_tokens` (non-cached input), `thoughts_tokens` / `response_tokens` (both the output price) and optional `cached_tokens` (cache-hit price). Values are stored in USD; pass `listSupportedModels("CNY")` to convert at 7 CNY/USD. + +The registry is the curated current line-up, so prefer it when picking a model or estimating cost. It is narrower than the routing rules: older ids in the table above (`gpt-5.4`, `claude-opus-4-7`, `gemini-3.1-pro-preview`, `gemini-3.1-flash-lite`) still route fine but no longer appear in it. + ## Routing and credentials -- Without `clientType`, the client auto-routes by model id substring: `gemini-3*`, `gemini-embedding`, `claude` 4-6/4-7/4-8, `gpt-5.4`/`gpt-5.5`, `glm-5`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint. +- Without `clientType`, the client auto-routes by model id substring, in this order: `gemini-3.6` / `gemini-3.5-flash-lite`, then `gemini-3*` / `gemini-embedding`, `claude` 4-6/4-7/4-8/-5, `gpt-5.4`/`gpt-5.5`, `glm-5.2`, `glm-5`, `kimi-k3`, `kimi-k2.5`/`kimi-k2.6`, `deepseek-v4`, `openai`+`embedding` (embeddings), `openai`. Ids matching none of these throw. The gateway variants in the table above hit the same substrings, so they route to the right family — just set `baseUrl` to the gateway endpoint. +- Exception: a Gemini id served by an OpenAI-compatible gateway (e.g. OpenRouter's `google/gemini-3.6-flash`) still matches the Gemini substring and would auto-route to the Google protocol client. Pass `clientType: "openai"` explicitly for those. - For any other OpenAI chat-completion compatible model (e.g. Qwen series via OpenRouter or SiliconFlow), pass `clientType: "openai"` plus `baseUrl` (embeddings endpoints use a different client type — see Embeddings below). - API key: constructor parameter first, then the provider environment variable — `DEEPSEEK_API_KEY`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`, `MOONSHOT_API_KEY`. Base URLs read the same names with `_BASE_URL`. @@ -87,9 +113,30 @@ for await (const event of client.streamingResponseStateful({ ``` - Each `event` is a `UniEvent`: `event_type` is `start` | `delta` | `stop`, and `content_items` carry the increments. -- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`) and `tools`. +- `config` accepts `max_tokens`, `temperature`, `system_prompt`, `thinking_level` (the `ThinkingLevel` enum, `NONE` to `XHIGH`), `tool_choice`, `prompt_caching` and `tools`. - `streamingResponseStateful` keeps conversation history inside the client; manage it with `getHistory()` / `setHistory(history)` / `clearHistory()`. The stateless variant is `streamingResponse({ messages, config })`. +## Config parameters the model may reject + +A config value the target client cannot honour throws `UnsupportedParameterError` (an `AgentHubError` carrying `client` and `parameter`) while building the request, before anything reaches the network: + +```ts +import { UnsupportedParameterError } from "@prismshadow/agenthub"; + +try { + // ... +} catch (err) { + if (err instanceof UnsupportedParameterError) console.error(err.parameter, err.message); +} +``` + +- `thinking_level` never throws: every client maps each level onto the closest one the model supports. Kimi K3 reasons unconditionally, so `NONE` degrades to its lowest effort rather than disabling thinking; GLM-5.2 sends `reasoning_effort` alongside its `thinking` block and only `NONE` disables it. +- `temperature` is rejected outright by Gemini 3.6 — that generation deprecated the sampling parameters, so the client refuses them instead of sending a value the API ignores. GPT-5.5, Claude 4.8/5, DeepSeek V4, Kimi K2.6 and Kimi K3 accept only the protocol default `1.0` and reject any other value. Gemini 3, Claude 4.6, GLM and the generic OpenAI client pass it through. +- `tool_choice`: `"auto"` is safe everywhere. Claude accepts a single forced tool name; DeepSeek V4 and Kimi K2.6 allow `"auto"` / `"none"`; Kimi K3 adds `"required"` but refuses a specific tool; GLM only accepts `"auto"`. +- `prompt_caching`: every client accepts `PromptCaching.ENABLE` and rejects the other values — caching is on by default and Kimi K3 caches context automatically. + +Leave a parameter unset and the protocol default applies, which is the portable choice when a script must run against several families. + ## Image generation Use a Gemini image model (see Model IDs) and set `config.image_config` (optional `aspect_ratio`, and `image_size` of `"1K"` | `"2K"`): @@ -97,7 +144,7 @@ Use a Gemini image model (see Model IDs) and set `config.image_config` (optional ```ts import fs from "node:fs"; -const client = new AutoLLMClient({ model: "gemini-3.1-flash-image-preview" }); +const client = new AutoLLMClient({ model: "gemini-3.1-flash-image" }); for await (const event of client.streamingResponseStateful({ message: { role: "user", content_items: [{ type: "text", text: "A penguin on a glacier" }] }, config: { image_config: { aspect_ratio: "16:9", image_size: "2K" } }, diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index a7e559c..4472e54 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -36,8 +36,8 @@ importers: packages/cli: dependencies: '@prismshadow/agenthub': - specifier: ^0.4.0 - version: 0.4.0(ws@8.21.0) + specifier: ^0.4.1 + version: 0.4.1(ws@8.21.0) '@prismshadow/penguin-core': specifier: workspace:* version: link:../core @@ -79,8 +79,8 @@ importers: packages/core: dependencies: '@prismshadow/agenthub': - specifier: ^0.4.0 - version: 0.4.0(ws@8.21.0) + specifier: ^0.4.1 + version: 0.4.1(ws@8.21.0) '@prismshadow/penguin-skills': specifier: workspace:* version: link:../skills @@ -1040,8 +1040,8 @@ packages: engines: {node: '>=18'} hasBin: true - '@prismshadow/agenthub@0.4.0': - resolution: {integrity: sha512-P96fAoYEtAy+BQHsrdQ2Rw5oXbrR2IAXcN6S3SnpUUlwpVoH2kkdtbe4bTGTafys0jJeH7s1lCZDF7KRvdWOoQ==} + '@prismshadow/agenthub@0.4.1': + resolution: {integrity: sha512-YKidCwa4ZO0+BP5E+vg/p8HQAPrTvTDaFmoj5omUHi7OxXCXPMDxwodj4VPz/ntWMDsgjsePhzkZIfVmSbECkw==} '@protobufjs/aspromise@1.1.2': resolution: {integrity: sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==} @@ -3678,7 +3678,7 @@ snapshots: dependencies: playwright: 1.61.1 - '@prismshadow/agenthub@0.4.0(ws@8.21.0)': + '@prismshadow/agenthub@0.4.1(ws@8.21.0)': dependencies: '@anthropic-ai/bedrock-sdk': 0.26.4 '@anthropic-ai/sdk': 0.81.0