diff --git a/packages/cli/src/commands/config.ts b/packages/cli/src/commands/config.ts index 64394bf..b1fdc45 100644 --- a/packages/cli/src/commands/config.ts +++ b/packages/cli/src/commands/config.ts @@ -2,7 +2,7 @@ * `penguin config` — manages a Project's model credentials, default model, model list, * Agent-level vault environment variables, and UI language. * - * penguin config model add --model-id --provider [--api-key ] [--context-window ] [--set-default] [--root ] + * penguin config model add --model-id --provider [--api-key ] [--context-window ] [--max-tokens ] [--set-default] [--root ] * penguin config model default --model-id --provider [--root ] * penguin config model vision --model-id --provider [--root ] * penguin config model list [--root ] @@ -120,6 +120,7 @@ export function registerConfigCommand(program: Command, t: Messages): void { .option("--api-key ", t.config.addApiKey) .option("--base-url ", t.config.addBaseUrl) .option("--context-window ", t.config.addContextWindow, parseIntArg) + .option("--max-tokens ", t.config.addMaxTokens, parseIntArg) .option("--client-type ", t.config.addClientType) // Tri-state: --vision marks it supported / --no-vision marks it unsupported / neither given keeps the existing value (defaults to supported). .option("--vision", t.config.addVision) @@ -131,6 +132,15 @@ export function registerConfigCommand(program: Command, t: Messages): void { .option("--set-default", t.config.addSetDefault, false) .option("--root ", t.common.root) .action(async (opts) => { + // Output cap: parseIntArg already rejects non-numbers; 0/negative must not reach the config either. + const maxTokens: number | undefined = opts.maxTokens; + if (maxTokens !== undefined && maxTokens <= 0) { + process.stderr.write( + `${t.error(`--max-tokens must be a positive integer: got "${maxTokens}".`)}\n`, + ); + process.exitCode = 1; + return; + } const root = resolveRootOption(opts.root); // --model-id takes the upstream id, paired with the required --provider as a // reference; the group is never guessed, so --api-key can only ever land on the @@ -164,6 +174,7 @@ export function registerConfigCommand(program: Command, t: Messages): void { provider, model_id: modelId, ...(opts.contextWindow !== undefined ? { context_window: opts.contextWindow } : {}), + ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}), ...(clientType !== undefined ? { client_type: clientType } : {}), ...(opts.vision !== undefined ? { vision: opts.vision } : {}), ...(Object.keys(pricing).length > 0 ? { pricing } : {}), diff --git a/packages/cli/src/i18n.ts b/packages/cli/src/i18n.ts index 094b6f8..8b24e05 100644 --- a/packages/cli/src/i18n.ts +++ b/packages/cli/src/i18n.ts @@ -39,6 +39,7 @@ export interface Messages { addApiKey: string; addBaseUrl: string; addContextWindow: string; + addMaxTokens: string; addClientType: string; addVision: string; addNoVision: string; @@ -174,6 +175,8 @@ const en: Messages = { addApiKey: "API key, stored inline in the Project's hidden .project_config.toml", addBaseUrl: "Custom base URL", addContextWindow: "Context window size (tokens)", + addMaxTokens: + "Per-model max output tokens (positive integer); when set it overrides the Agent's max_tokens, omit to inherit — lower it for small-context models", addClientType: "AgentHub client type (e.g. openai); defaults by provider group when omitted", addVision: "Mark the model as supporting image input (vision)", addNoVision: "Mark the model as NOT supporting image input; omit both to keep current", @@ -287,6 +290,8 @@ const zh: Messages = { addApiKey: "API key,内联存入 Project 的隐藏文件 .project_config.toml", addBaseUrl: "自定义 base url", addContextWindow: "上下文窗口大小(token 数)", + addMaxTokens: + "该模型的最大输出长度(正整数);设置后覆盖 Agent 的 max_tokens,缺省沿用——小上下文模型建议调低", addClientType: "AgentHub 客户端协议(如 openai);缺省按 provider 分组的语义取值", addVision: "标注该模型支持图片输入(视觉)", addNoVision: "标注该模型不支持图片输入;两者都不给则保留原值", diff --git a/packages/cli/test/config-model.test.ts b/packages/cli/test/config-model.test.ts index b3f13ee..7fe50bc 100644 --- a/packages/cli/test/config-model.test.ts +++ b/packages/cli/test/config-model.test.ts @@ -229,6 +229,76 @@ describe("penguin config model add/list (--root plus provider / model_id stored expect(by("mylab", "special-1").client_type).toBe("verbatim-type"); }); + it("--max-tokens round-trips to the entry's max_tokens; 0/negative/non-number are rejected before anything is written", async () => { + const add = await runModel([ + "add", + "--model-id", + "local-32k", + "--provider", + "custom", + "--base-url", + "http://127.0.0.1:8000/v1", + "--max-tokens", + "8000", + "--root", + tmpRoot, + ]); + expect(add.code).toBe(0); + const entryOf = async () => { + const parsed = parseToml( + await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"), + ) as { models: Array> }; + return parsed.models.find((m) => m.provider === "custom" && m.model_id === "local-32k"); + }; + expect((await entryOf())?.max_tokens).toBe(8000); + + // Upsert without --max-tokens keeps the existing annotation (same merge policy as context_window). + const update = await runModel([ + "add", + "--model-id", + "local-32k", + "--provider", + "custom", + "--context-window", + "32768", + "--root", + tmpRoot, + ]); + expect(update.code).toBe(0); + expect((await entryOf())?.max_tokens).toBe(8000); + + // 0 / negative: rejected with a clear error, and the config is untouched. + for (const bad of ["0", "-5"]) { + const res = await runModel([ + "add", + "--model-id", + "local-32k", + "--provider", + "custom", + "--max-tokens", + bad, + "--root", + tmpRoot, + ]); + expect(res.code).toBe(1); + expect(res.err).toContain("--max-tokens must be a positive integer"); + } + // Non-number: parseIntArg throws a commander usage error (nonzero exit). + const nan = await runModel([ + "add", + "--model-id", + "local-32k", + "--provider", + "custom", + "--max-tokens", + "many", + "--root", + tmpRoot, + ]); + expect(nan.code).not.toBe(0); + expect((await entryOf())?.max_tokens).toBe(8000); + }); + it("model default sets the default model under the --root data root (--model-id upstream id + --provider as a pair)", async () => { const set = await runModel([ "default", diff --git a/packages/core/src/agent.ts b/packages/core/src/agent.ts index 63db090..7d0b6b8 100644 --- a/packages/core/src/agent.ts +++ b/packages/core/src/agent.ts @@ -117,6 +117,16 @@ export function effectiveMaxContextLength(configured: number, contextWindow: unk return Math.min(configured, Math.floor(contextWindow * 0.75)); } +/** + * Output cap for meta requests (title generation / vision describing): these carry their own + * small hardcoded budget, tightened further by the entry's per-model `max_tokens` when that is + * smaller — a cap the user pinned below the budget must bind every request to that model. The + * budget is never raised. + */ +export function metaMaxTokens(budget: number, modelCap: number | undefined): number { + return modelCap !== undefined ? Math.min(budget, modelCap) : budget; +} + /** Create or load an Agent. */ export async function createAgent(opts: CreateAgentOptions = {}): Promise { const state = await loadOrInitAgentState(opts); @@ -570,7 +580,8 @@ export class Agent { : {}), tools: [], thinkingLevel: "none", - maxTokens: 2048, + // The describing budget, tightened by the vision entry's own pinned cap when smaller. + maxTokens: metaMaxTokens(2048, visionEntry.max_tokens), requestTimeoutMs: 60_000, }), }; @@ -592,6 +603,12 @@ export class Agent { }); const tools = await environment.listTools(); + // Effective output cap: the entry's per-model annotation wins over the Agent's + // system_config value — the fit is a model trait: the seeded per-Agent default (32000) + // cannot fit into e.g. a 32768-token context window together with any prompt, so a + // small-window model needs its own pinned cap. Unset inherits the Agent value. + const maxTokens = modelEntry.max_tokens ?? this.state.systemConfig.model?.max_tokens; + // LLM constructor args are extracted into a constant so they can be reused as-is when // rebuilding a new LLM object after compaction (with a fresh model context) — the system // prompt and tool definitions aren't part of the compacted history, so the new object keeps @@ -612,9 +629,7 @@ export class Agent { ...(modelEntry.context_window !== undefined ? { contextWindow: modelEntry.context_window } : {}), - ...(this.state.systemConfig.model?.max_tokens !== undefined - ? { maxTokens: this.state.systemConfig.model.max_tokens } - : {}), + ...(maxTokens !== undefined ? { maxTokens } : {}), ...(this.state.systemConfig.model?.thinking_level !== undefined ? { thinkingLevel: this.state.systemConfig.model.thinking_level } : {}), @@ -640,7 +655,8 @@ export class Agent { ...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}), tools: [], thinkingLevel: "none", - maxTokens: 300, + // The meta budget, tightened by the entry's pinned per-model cap when smaller. + maxTokens: metaMaxTokens(300, modelEntry.max_tokens), requestTimeoutMs: 30_000, }); diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 430d953..76211ad 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -40,13 +40,13 @@ export type { } from "./engine/context-engine.js"; export { Session } from "./session.js"; export type { SessionConfig } from "./session.js"; -export { - buildTitlePrompt, - generateTitleWithLLM, - sanitizeTitle, - stripConversationMarkers, -} from "./session-title.js"; -export type { SessionTitleResult } from "./session-title.js"; +// Session-title generation lives in internal/ (an assembly detail of Session.generateTitle); +// only its narrow public surface is re-exported: the result type (part of +// Session.generateTitle's signature) and the sanitation helpers the Web server's title +// fallback builds on (stripConversationMarkers / sanitizeTitle). The prompt/request +// internals (buildTitlePrompt / generateTitleWithLLM) are deliberately not public. +export { sanitizeTitle, stripConversationMarkers } from "./internal/session-title.js"; +export type { SessionTitleResult } from "./internal/session-title.js"; export { Agent, createAgent } from "./agent.js"; export type { CreateAgentOptions, CreateSessionOptions, ResumeSessionOptions } from "./agent.js"; diff --git a/packages/core/src/session-title.ts b/packages/core/src/internal/session-title.ts similarity index 92% rename from packages/core/src/session-title.ts rename to packages/core/src/internal/session-title.ts index cf99e22..bc6515b 100644 --- a/packages/core/src/session-title.ts +++ b/packages/core/src/internal/session-title.ts @@ -1,21 +1,25 @@ /** * Session title generation: an **out-of-band, one-off request** that generates - * a short title from the first-turn conversation text. + * a short title from the first-turn conversation text (used by `Session.generateTitle` + * for assembly, not exported wholesale via the barrel). * * Called by `session.generateTitle()`: sends one request using the bare LLM for the session's * Model (no tools, no system prompt, thinking off), without writing history or Trace. Material * defaults to what the Session self-captures during run (see session.ts); this module is only * responsible for the prompt format, driving the one-off request, and sanitizing the result — * when to generate a title and where to store it is decided by the host (Web server / CLI). + * The narrow public surface — `SessionTitleResult` (part of `Session.generateTitle`'s + * signature) and the sanitation helpers the host's title fallback builds on — is re-exported + * by the barrel; the prompt/request internals are not. */ -import { userText } from "./omnimessage/index.js"; +import { userText } from "../omnimessage/index.js"; import type { OmniMessage, TextPayload, TokenCounts, TokenUsagePayload, -} from "./omnimessage/index.js"; -import type { LLMInterface } from "./interfaces.js"; +} from "../omnimessage/index.js"; +import type { LLMInterface } from "../interfaces.js"; /** Cap on conversation text spliced into the title request (user/model each truncated separately, to control cost). */ const EXCERPT_MAX_CHARS = 2000; diff --git a/packages/core/src/session.ts b/packages/core/src/session.ts index c30a2f9..f273ac6 100644 --- a/packages/core/src/session.ts +++ b/packages/core/src/session.ts @@ -21,8 +21,8 @@ import { sessionMeta } from "./omnimessage/index.js"; import type { OmniMessage, SessionMetaPayload, TokenCounts } from "./omnimessage/index.js"; import { imagesToScratchpadPaths } from "./internal/session-support.js"; import type { EnvironmentInterface, LLMInterface, ToolPermission } from "./interfaces.js"; -import { generateTitleWithLLM } from "./session-title.js"; -import type { SessionTitleResult } from "./session-title.js"; +import { generateTitleWithLLM } from "./internal/session-title.js"; +import type { SessionTitleResult } from "./internal/session-title.js"; import { ContextEngine } from "./engine/context-engine.js"; import type { CompactAvailability, diff --git a/packages/core/src/state/project-config.ts b/packages/core/src/state/project-config.ts index d0f4997..aeb9896 100644 --- a/packages/core/src/state/project-config.ts +++ b/packages/core/src/state/project-config.ts @@ -88,6 +88,14 @@ export interface ModelEntry { * enters that session's history. */ vision?: boolean; + /** + * Per-model max output tokens (the request's output cap, i.e. GenerativeModelConfig.maxTokens): + * when set it wins over the Agent's `system_config.model.max_tokens` — the fit is a model trait + * (the seeded per-Agent default of 32000 cannot fit into e.g. a 32768-token context window + * together with any prompt, and the upstream rejects the request outright). Unset = inherit + * the Agent value. User-only, never preset by the builtin catalog. + */ + max_tokens?: number; /** Pricing info; absent means this Model's cost isn't counted. */ pricing?: ModelPricing; /** API key (inlined credential); left empty falls back to the vendor's environment variable. */ @@ -275,6 +283,8 @@ export async function addModel( client_type?: string; /** Whether image input is supported (vision/multimodal); keeps the existing value by default (treated as supported if never set). */ vision?: boolean; + /** Per-model max output tokens (wins over the Agent config); keeps the existing value by default (unset = inherit the Agent value). */ + max_tokens?: number; /** Price input may cover only some buckets; merged and written as a complete `ModelPricing`. */ pricing?: Partial; api_key?: string; @@ -310,6 +320,10 @@ export async function addModel( if (vision !== undefined) { modelEntry.vision = vision; } + const maxTokens = entry.max_tokens ?? existing?.max_tokens; + if (maxTokens !== undefined) { + modelEntry.max_tokens = maxTokens; + } // The three price buckets are merged field by field: an unspecified bucket keeps its existing // value (the same policy as context_window/credential); the unit is fixed to usd_per_mtok, and // the complete pricing is written as long as any bucket is present. diff --git a/packages/core/test/agent.test.ts b/packages/core/test/agent.test.ts index d332bf5..5539e96 100644 --- a/packages/core/test/agent.test.ts +++ b/packages/core/test/agent.test.ts @@ -22,7 +22,7 @@ import { installSkill, setVaultEntry, } from "../src/index.js"; -import { effectiveMaxContextLength } from "../src/agent.js"; +import { effectiveMaxContextLength, metaMaxTokens } from "../src/agent.js"; import { stubProviderKeys } from "./provider-keys.js"; let tmpRoot: string; @@ -53,6 +53,15 @@ describe("effectiveMaxContextLength (compaction threshold clamped to the model w }); }); +describe("metaMaxTokens (meta-request budget tightened by the per-model cap)", () => { + it("keeps the budget unless the per-model cap is smaller; never raises it", () => { + expect(metaMaxTokens(300, undefined)).toBe(300); // no per-model cap: the budget as-is + expect(metaMaxTokens(300, 8000)).toBe(300); // ample cap: the small budget stays + expect(metaMaxTokens(300, 128)).toBe(128); // pinned below the budget: the cap binds + expect(metaMaxTokens(2048, 1024)).toBe(1024); // vision-describer budget, same rule + }); +}); + describe("Agent.createSession workspace handling", () => { it("throws a clear error when the given workspace does not exist (no auto-create)", async () => { const agent = await createAgent(); @@ -192,6 +201,93 @@ describe("Agent.createSession model reference ((provider, model_id) pair)", () = }); }); +describe("Agent.createSession max output tokens (per-model cap wins over the Agent config)", () => { + // Reads a constructed GenerativeModel's request config (private; runtime-accessible for assertion). + const uniConfigOf = (llm: unknown) => + (llm as { uniConfig?: { max_tokens?: number } }).uniConfig ?? {}; + + it("uses the entry's max_tokens in llmConfig, and inherits the seeded 32000 when unset", async () => { + // A 32k-context local model: the seeded per-Agent default (32000 output tokens) cannot fit + // into its window together with any prompt — the pinned per-model cap must win. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "local-32k", + client_type: "openai", + context_window: 32768, + max_tokens: 8000, + }); + const agent = await createAgent(); + expect(agent.state.systemConfig.model?.max_tokens).toBe(32000); + const ws = path.join(tmpRoot, "ws-max-tokens"); + await fs.mkdir(ws, { recursive: true }); + + const pinned = await agent.createSession({ + workspaceDir: ws, + modelId: "local-32k", + provider: "custom", + }); + try { + const llm = (pinned as unknown as { engine: { deps: { llm: unknown } } }).engine.deps.llm; + expect(uniConfigOf(llm).max_tokens).toBe(8000); + } finally { + pinned.dispose(); + } + + // An unannotated entry (the default model) inherits the Agent value, as before. + const inherited = await agent.createSession({ workspaceDir: ws }); + try { + const llm = (inherited as unknown as { engine: { deps: { llm: unknown } } }).engine.deps.llm; + expect(uniConfigOf(llm).max_tokens).toBe(32000); + } finally { + inherited.dispose(); + } + }); + + it("meta requests keep their small budget, tightened when the per-model cap is even smaller", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "local-32k", + client_type: "openai", + max_tokens: 8000, + }); + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "tiny-cap", + client_type: "openai", + max_tokens: 128, + }); + const agent = await createAgent(); + const ws = path.join(tmpRoot, "ws-meta-cap"); + await fs.mkdir(ws, { recursive: true }); + + // Ample per-model cap: the title one-shot keeps its own 300 budget (never raised to the cap). + const ample = await agent.createSession({ + workspaceDir: ws, + modelId: "local-32k", + provider: "custom", + }); + try { + const bare = (ample as unknown as { createBareLLM?: () => unknown }).createBareLLM?.(); + expect(uniConfigOf(bare).max_tokens).toBe(300); + } finally { + ample.dispose(); + } + + // Cap pinned below the budget: the meta request must respect it too. + const tiny = await agent.createSession({ + workspaceDir: ws, + modelId: "tiny-cap", + provider: "custom", + }); + try { + const bare = (tiny as unknown as { createBareLLM?: () => unknown }).createBareLLM?.(); + expect(uniConfigOf(bare).max_tokens).toBe(128); + } finally { + tiny.dispose(); + } + }); +}); + describe("Agent.createSession vault injection", () => { it("injects vault key names (never values) into the assembled system prompt", async () => { // Write the Agent vault to disk first; createSession reads that Agent's own .vault.toml. diff --git a/packages/core/test/session-title.test.ts b/packages/core/test/session-title.test.ts index 2ef18aa..13ce130 100644 --- a/packages/core/test/session-title.test.ts +++ b/packages/core/test/session-title.test.ts @@ -5,9 +5,7 @@ import { describe, it, expect } from "vitest"; import { assistantText, - buildTitlePrompt, emptyTokenCounts, - generateTitleWithLLM, sanitizeTitle, Session, stripConversationMarkers, @@ -15,6 +13,8 @@ import { tokenUsage, userText, } from "../src/index.js"; +// Prompt/request internals are no longer exported via the barrel: imported directly from the internal module. +import { buildTitlePrompt, generateTitleWithLLM } from "../src/internal/session-title.js"; import type { EnvironmentInterface, LLMInterface, diff --git a/packages/core/test/state.test.ts b/packages/core/test/state.test.ts index 32365cb..804bfc4 100644 --- a/packages/core/test/state.test.ts +++ b/packages/core/test/state.test.ts @@ -531,6 +531,7 @@ describe("project-config round trip", () => { provider: "custom", model_id: "gpt-test", context_window: 128000, + max_tokens: 8192, api_key: "sk-abc", base_url: "https://example.com/v1", }, @@ -549,6 +550,7 @@ describe("project-config round trip", () => { provider: "custom", model_id: "gpt-test", context_window: 128000, + max_tokens: 8192, api_key: "sk-abc", base_url: "https://example.com/v1", }); @@ -644,6 +646,32 @@ describe("project-config round trip", () => { expect(m?.vision).toBe(true); }); + it("addModel persists max_tokens and upsert preserves it when not re-specified", async () => { + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "small-window", + max_tokens: 4096, + }); + // Only supplements context_window, without max_tokens: the original annotation is kept. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "small-window", + context_window: 32768, + }); + const ref = { provider: "custom", model_id: "small-window" }; + let m = getModel(await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID), ref); + expect(m?.max_tokens).toBe(4096); + expect(m?.context_window).toBe(32768); + // Explicitly re-pins the cap. + await addModel(tmpRoot, DEFAULT_PROJECT_ID, { + provider: "custom", + model_id: "small-window", + max_tokens: 2048, + }); + m = getModel(await loadProjectConfig(tmpRoot, DEFAULT_PROJECT_ID), ref); + expect(m?.max_tokens).toBe(2048); + }); + it("setVisionModel persists and validates the target", async () => { await addModel(tmpRoot, DEFAULT_PROJECT_ID, { provider: "custom", diff --git a/packages/docs/content/cli.en.md b/packages/docs/content/cli.en.md index 6d2a674..730c75c 100644 --- a/packages/docs/content/cli.en.md +++ b/packages/docs/content/cli.en.md @@ -84,6 +84,7 @@ penguin config model add --provider deepseek --model-id deepseek-v4-pro --api-ke | `--api-key ` | API key, stored inline in the Project's hidden `.project_config.toml` | | `--base-url ` | Custom endpoint base URL | | `--context-window ` | Context window size | +| `--max-tokens ` | Per-model max output tokens (positive integer). Overrides the Agent's `model.max_tokens` when set; omit to inherit — lower it for small-context models | | `--client-type ` | Client protocol type | | `--vision` / `--no-vision` | Mark vision input as supported / unsupported | | `--price-cache-read ` | Cache-read price | diff --git a/packages/docs/content/cli.zh.md b/packages/docs/content/cli.zh.md index 163b9f8..7fcc81c 100644 --- a/packages/docs/content/cli.zh.md +++ b/packages/docs/content/cli.zh.md @@ -84,6 +84,7 @@ penguin config model add --provider deepseek --model-id deepseek-v4-pro --api-ke | `--api-key ` | API Key,内联存入 Project 隐藏文件 `.project_config.toml` | | `--base-url ` | 自定义接口地址 | | `--context-window ` | 上下文窗口大小 | +| `--max-tokens ` | 该模型的最大输出长度(正整数)。设置后覆盖 Agent 的 `model.max_tokens`,缺省沿用;小上下文模型建议调低 | | `--client-type ` | 客户端协议类型 | | `--vision` / `--no-vision` | 标记是否支持视觉输入 | | `--price-cache-read ` | 缓存读价格 | diff --git a/packages/docs/content/models.en.md b/packages/docs/content/models.en.md index 64ba372..4fb5e61 100644 --- a/packages/docs/content/models.en.md +++ b/packages/docs/content/models.en.md @@ -22,6 +22,7 @@ Each Project's available models are recorded in the hidden `.project_config.toml | `provider` | Config group name; paired with `model_id` it forms the unique key | | `model_id` | Upstream request id | | `context_window` | Context window | +| `max_tokens` | Optional per-model output cap (max output tokens per request). When set it overrides the Agent's `model.max_tokens`; unset inherits it. Lower it for small-context models: the per-Agent default (32000) cannot fit into e.g. a 32k window together with any prompt. Omitting the field on a Web full-table save clears it | | `client_type` | Protocol hint (e.g. `openai`); inferred by AgentHub from the model id when omitted | | `display_name` | Display name | | `vision` | Whether image input is supported, default true | @@ -76,7 +77,7 @@ Some models in the preset catalog: deepseek-v4-pro / deepseek-v4-flash, gemini-3 ## Thinking levels -Five levels: `none | low | medium | high | xhigh`, configured per Agent as `model.thinking_level` in `system_config.yaml`, default medium. See [Configuration](/configuration). +Five levels: `none | low | medium | high | xhigh`, configured per Agent as `model.thinking_level` in `system_config.yaml`, default medium. The chat draft view offers a quick picker next to the model selector: a picked level is written back to the selected Agent's setting immediately (the switched-to level becomes that Agent's new default and applies from the next session; a running session keeps the level it was created with). See [Configuration](/configuration). ## Models decoupled from Agents diff --git a/packages/docs/content/models.zh.md b/packages/docs/content/models.zh.md index ef8eacc..e296e56 100644 --- a/packages/docs/content/models.zh.md +++ b/packages/docs/content/models.zh.md @@ -22,6 +22,7 @@ description: 经 AgentHub 单一网关接入模型,以 (provider, model_id) | `provider` | 配置分组名,与 `model_id` 成对构成唯一键 | | `model_id` | 上游请求 id | | `context_window` | 上下文窗口 | +| `max_tokens` | 可选的按模型输出上限(单次请求最大输出 Token 数)。设置后覆盖 Agent 的 `model.max_tokens`,缺省沿用;小上下文模型建议调低——按 Agent 的缺省值(32000)加上任意 Prompt 无法放进如 32k 的窗口。Web 整表保存时省略该字段即清除 | | `client_type` | 协议提示(如 `openai`);缺省由 AgentHub 按 model id 推断 | | `display_name` | 显示名 | | `vision` | 是否支持图像输入,默认 true | @@ -76,7 +77,7 @@ api_key = "sk-..." ## 思考等级 -思考等级共五档:`none | low | medium | high | xhigh`,按 Agent 在 `system_config.yaml` 的 `model.thinking_level` 配置,默认 medium。见 [配置参考](/configuration)。 +思考等级共五档:`none | low | medium | high | xhigh`,按 Agent 在 `system_config.yaml` 的 `model.thinking_level` 配置,默认 medium。对话草稿页在模型选择器旁提供快捷拾取器:选定档位立即写回所选 Agent 的该项配置(切换后的档位即成为该 Agent 的新默认,自下一个 Session 生效;进行中的 Session 沿用创建时的档位)。见 [配置参考](/configuration)。 ## 模型与 Agent 解耦 diff --git a/packages/docs/content/web-app.en.md b/packages/docs/content/web-app.en.md index 01f9572..42ee302 100644 --- a/packages/docs/content/web-app.en.md +++ b/packages/docs/content/web-app.en.md @@ -33,7 +33,7 @@ The interface language (中文 / English / system) and theme (light / dark / sys ### Creating a Conversation -A new conversation starts as a draft: pick the Agent, the Workspace (via a server-side directory browser), the approval mode, and the model before sending the first message. The Session is created on first send, and from then on its model and Workspace are locked. +A new conversation starts as a draft: pick the Agent, the Workspace (via a server-side directory browser), the approval mode, the model, and the thinking level before sending the first message. The Session is created on first send, and from then on its model and Workspace are locked. Switching the thinking level or the model makes the switched-to value the new default: the level is written back to the selected Agent's `model.thinking_level` immediately, and the picked model carries over as the next conversation's default; a running session keeps the level and model it was created with (the input area shows them read-only). There are four approval modes: `allow-all`, `deny-all`, `read-only` (only read-only tools pass), and `always-ask`. See [Tools and Approvals](/tools). @@ -76,7 +76,7 @@ Browse the Skill library by group, install Skills onto an Agent, or quick-invoke ## Model Configuration (/models) -A per-Project model table grouped by provider. Models can be added and edited: identity is the `(provider, model_id)` pair, credentials are masked, and context window, pricing, and the vision flag are configurable. You can set the default model and the vision model (which reads images on behalf of session models without image input), and run a connectivity test on any entry. Only Project owners can edit. For concepts, see [Models and Providers](/models). +A per-Project model table grouped by provider. Models can be added and edited: identity is the `(provider, model_id)` pair, credentials are masked, and context window, max output tokens (a per-model cap overriding the Agent's `model.max_tokens` — lower it for small-context models), pricing, and the vision flag are configurable. You can set the default model and the vision model (which reads images on behalf of session models without image input), and run a connectivity test on any entry. Only Project owners can edit. For concepts, see [Models and Providers](/models). ## Usage (/usage) diff --git a/packages/docs/content/web-app.zh.md b/packages/docs/content/web-app.zh.md index 989ace0..aeafb25 100644 --- a/packages/docs/content/web-app.zh.md +++ b/packages/docs/content/web-app.zh.md @@ -33,7 +33,7 @@ penguin web ### 新建会话 -新会话从草稿开始:先选择 Agent、Workspace(服务器端目录浏览器选取)、审批模式与模型,再发送第一条消息。Session 在首次发送时才真正创建,此后该会话的模型与 Workspace 即被锁定。 +新会话从草稿开始:先选择 Agent、Workspace(服务器端目录浏览器选取)、审批模式、模型与思考等级,再发送第一条消息。Session 在首次发送时才真正创建,此后该会话的模型与 Workspace 即被锁定。切换思考等级或模型时,切换后的值即成为新的默认:思考等级立即写回所选 Agent 的 `model.thinking_level`,所选模型则作为下一个新会话的默认延续;进行中的会话沿用创建时的档位与模型(输入区只读展示)。 审批模式共四种:`allow-all`(全部放行)、`deny-all`(全部拒绝)、`read-only`(仅放行只读工具)、`always-ask`(每次询问),详见[工具与审批](/tools)。 @@ -76,7 +76,7 @@ penguin web ## 模型配置(/models) -按 Provider 分组展示当前 Project 的模型表格。支持添加与编辑模型:以 `(provider, model_id)` 为唯一标识,凭据以掩码显示,可配置上下文窗口、定价与视觉(vision)标记;可设置默认模型与视觉模型(在会话模型不支持图片输入时代为读图),并对任一模型做连通性测试。仅 Project Owner 可编辑,概念说明见[模型与 Provider](/models)。 +按 Provider 分组展示当前 Project 的模型表格。支持添加与编辑模型:以 `(provider, model_id)` 为唯一标识,凭据以掩码显示,可配置上下文窗口、最大输出长度(按模型的输出上限,覆盖 Agent 的 `model.max_tokens`——小上下文模型建议调低)、定价与视觉(vision)标记;可设置默认模型与视觉模型(在会话模型不支持图片输入时代为读图),并对任一模型做连通性测试。仅 Project Owner 可编辑,概念说明见[模型与 Provider](/models)。 ## 用量统计(/usage) diff --git a/packages/server/src/api/types.ts b/packages/server/src/api/types.ts index 47f8ac6..710be5d 100644 --- a/packages/server/src/api/types.ts +++ b/packages/server/src/api/types.ts @@ -206,6 +206,13 @@ export interface ModelInfo { * unset (= treated as supported). */ vision?: boolean; + /** + * Per-model max output tokens (TOML `max_tokens` annotation; user-only, never preset by the + * built-in catalog): when set it wins over the Agent's `system_config.model.max_tokens`; + * unset = inherit the Agent value. Lets a small-context model cap its output below the + * seeded per-Agent default (32000), which cannot fit into e.g. a 32k context window. + */ + maxTokens?: number; pricing?: ModelPricingDto; /** Environment variable name to fall back to when api_key is empty (e.g. ANTHROPIC_API_KEY); unset if no known fallback. */ envKey?: string; @@ -241,6 +248,8 @@ export interface ModelUpdateEntry { clientType?: string; /** Whether image input (vision/multimodal) is supported; omitted = supported (not persisted). */ vision?: boolean; + /** Per-model max output tokens, a positive integer (wins over the Agent config); omitted = inherit the Agent value (the annotation is cleared). */ + maxTokens?: number; pricing?: ModelPricingDto; /** Providing it overwrites and updates createdAt; omitting it keeps the existing value. */ apiKey?: string; diff --git a/packages/server/src/http/routes/models.ts b/packages/server/src/http/routes/models.ts index d197797..722c38a 100644 --- a/packages/server/src/http/routes/models.ts +++ b/packages/server/src/http/routes/models.ts @@ -76,6 +76,12 @@ function parseModelsUpdate(body: Record): ModelsUpdateRequest { } entry.vision = m.vision; } + if (m.maxTokens !== undefined) { + if (typeof m.maxTokens !== "number" || !Number.isInteger(m.maxTokens) || m.maxTokens <= 0) { + throw badRequest(`models[${i}].maxTokens must be a positive integer.`); + } + entry.maxTokens = m.maxTokens; + } if (m.pricing !== undefined) { const p = m.pricing as Record; if (p === null || typeof p !== "object" || Array.isArray(p)) { diff --git a/packages/server/src/services/project-config-service.ts b/packages/server/src/services/project-config-service.ts index 7779e68..f2f2c32 100644 --- a/packages/server/src/services/project-config-service.ts +++ b/packages/server/src/services/project-config-service.ts @@ -350,6 +350,8 @@ export class ProjectConfigService { // routed has no fallback (no envKey, and AgentHub will reject that id). const envKey = resolveModelEnv(modelId, clientType)?.envKey; const vision = typeof m.vision === "boolean" ? m.vision : cat?.supportsVision; + // Output cap: TOML annotation only (user-owned; the built-in catalog never presets it). + const maxTokens = optNum(m.max_tokens); // Display name: the explicit TOML field (user-edited) takes priority, then the built-in catalog. const displayName = optStr(m.display_name) ?? cat?.displayName; // credential is inlined on the entry: a credential block is emitted if either api_key or base_url is present. @@ -369,6 +371,7 @@ export class ProjectConfigService { : {}), ...(clientType ? { clientType } : {}), ...(vision !== undefined ? { vision } : {}), + ...(maxTokens !== undefined ? { maxTokens } : {}), ...(envKey ? { envKey } : {}), ...(pricingDto ? { pricing: pricingDto } : {}), ...(apiKey !== undefined || credBaseUrl !== undefined @@ -443,6 +446,7 @@ export class ProjectConfigService { delete next.context_window; delete next.client_type; delete next.vision; + delete next.max_tokens; delete next.pricing; delete next.display_name; // Leftover key from the old concatenated format (request_model_id): defensively stripped, never written to disk again. @@ -460,6 +464,8 @@ export class ProjectConfigService { if (entry.clientType) next.client_type = entry.clientType; // Treated as supported by default: only written to disk when explicitly annotated (both true/false are kept; false drives a frontend blocking hint). if (entry.vision !== undefined) next.vision = entry.vision; + // Inherit-the-Agent-value by default: only written to disk when explicitly annotated (omitted on a full-table PUT = the annotation is cleared). + if (entry.maxTokens !== undefined) next.max_tokens = entry.maxTokens; if (entry.pricing !== undefined) { next.pricing = { unit: "usd_per_mtok", diff --git a/packages/server/test/models.test.ts b/packages/server/test/models.test.ts index a21b366..9340c20 100644 --- a/packages/server/test/models.test.ts +++ b/packages/server/test/models.test.ts @@ -162,6 +162,43 @@ describe("models preset & catalog enrichment", () => { expect(noProvider.status).toBe(400); }); + it("PUT maxTokens:落盘为 max_tokens 并经 GET 回读;整表省略即清除;0/负数/非数字 400", async () => { + const put = await api.put(url(), { + models: [ + { provider: "custom", modelId: "local-qwen", clientType: "openai", maxTokens: 8000 }, + ], + }); + expect(put.status).toBe(200); + expect(pick((await put.json()) as ModelsResponse, "custom", "local-qwen").maxTokens).toBe(8000); + + // Round-trips through disk (persisted as snake_case on the entry, not just echoed back). + const again = (await (await api.get(url())).json()) as ModelsResponse; + expect(pick(again, "custom", "local-qwen").maxTokens).toBe(8000); + const toml = await readFile(path.join(t.root, projectId, ".project_config.toml"), "utf8"); + expect(toml).toContain("max_tokens = 8000"); + + // Full-table PUT omitting the field clears the annotation (same replace semantics as vision/contextWindow). + const cleared = await api.put(url(), { + models: [{ provider: "custom", modelId: "local-qwen", clientType: "openai" }], + }); + expect(cleared.status).toBe(200); + const clearedRow = pick((await cleared.json()) as ModelsResponse, "custom", "local-qwen"); + expect("maxTokens" in clearedRow).toBe(false); + expect( + await readFile(path.join(t.root, projectId, ".project_config.toml"), "utf8"), + ).not.toContain("max_tokens"); + + // Not a positive integer → 400 with the field-labelled message (nothing written). + for (const bad of [0, -5, 1.5, "8000"]) { + const res = await api.put(url(), { + models: [{ provider: "custom", modelId: "local-qwen", maxTokens: bad }], + }); + expect(res.status).toBe(400); + const body = (await res.json()) as { error: { message: string } }; + expect(body.error.message).toContain("models[0].maxTokens"); + } + }); + it("同名 model_id 可在不同 provider 下并存(成对键,互不覆盖)", async () => { const put = await api.put(url(), { models: [ diff --git a/packages/skills/skills/penguin-cli/SKILL.md b/packages/skills/skills/penguin-cli/SKILL.md index 6a536a4..3c77017 100644 --- a/packages/skills/skills/penguin-cli/SKILL.md +++ b/packages/skills/skills/penguin-cli/SKILL.md @@ -3,7 +3,7 @@ name: penguin-cli description: Manage model API keys, default models and per-agent vault secrets with the penguin CLI. short_description: Manage models and secrets with the penguin CLI. short_description_zh: 用 penguin CLI 管理模型与密钥。 -version: 4 +version: 5 updated: 2026-07-22T00:00:00Z --- @@ -21,7 +21,7 @@ Add or update a model (upsert by the `(provider, model_id)` pair; re-run with mo ```bash penguin config model add --provider --model-id [--api-key ] [--base-url ] \ - [--client-type ] [--context-window ] [--vision | --no-vision] \ + [--client-type ] [--context-window ] [--max-tokens ] [--vision | --no-vision] \ [--price-cache-read ] [--price-cache-write ] [--price-output ] \ [--project-id ] [--root ] [--set-default] ``` @@ -30,6 +30,7 @@ penguin config model add --provider --model-id [--api-key - For any OpenAI chat-completion compatible endpoint use `--client-type openai --base-url `; omit `--client-type` to auto-route by model id. - Prices are USD per million tokens (cache read / cache write / output). - `--vision` / `--no-vision` mark whether the model accepts images; omitting both keeps the current value (default is vision-capable). +- `--max-tokens ` pins a per-model output cap (positive integer), overriding the Agent's `model.max_tokens`; omit to inherit. Lower it for small-context models — the per-Agent default (32000) cannot fit into e.g. a 32k context window together with any prompt. - All `penguin config model ...` and `penguin config vault ...` commands accept `--root ` to target another data root (default `PENGUIN_HOME`, then `~/.penguin/data`). Two configuration targets — treat the difference as a hard rule: - **Penguin's own model** (self-configuration: the model Penguin itself runs on): the default root without `--root` is correct. - **An AI app you are building**: `--root` **must** point at the app's own data directory inside the project (e.g. `--root ./penguin_data`, the same path the app gives `createAgent({ root })`) unless the user explicitly chose another location — never write an app's models or keys into the global `~/.penguin/data`, which belongs to the person running Penguin, not to the app. diff --git a/packages/web/e2e/draft.spec.mjs b/packages/web/e2e/draft.spec.mjs index 9ebfa24..d127361 100644 --- a/packages/web/e2e/draft.spec.mjs +++ b/packages/web/e2e/draft.spec.mjs @@ -86,6 +86,26 @@ test("draft: pick model/approval -> reload restores them -> send creates the ses await page.getByRole("button", { name: /放行只读/ }).click(); await expect(page.getByRole("button", { name: "审批模式" })).toContainText("放行只读"); + // Conversation-time thinking level (backed by the Agent settings): the picker shows the + // seeded default (medium, short name 中); the menu carries a title bar and exactly the five + // short-name rows (no descriptions, no default row); picking 高 writes straight through to + // the Agent config, so the session created on send runs with it and it becomes the Agent's + // new default. + const thinkingBtn = page.getByRole("button", { name: "思考等级" }); + await expect(thinkingBtn).toContainText("中"); + await thinkingBtn.click(); + await expect(page.getByText("思考等级", { exact: true })).toBeVisible(); // menu title bar + await page.getByRole("button", { name: "高", exact: true }).click(); + await expect(thinkingBtn).toContainText("高"); + await expect + .poll(async () => { + const cfg = await ( + await page.request.get(`${BASE}/api/projects/${projectId}/agents/default_agent/config`) + ).json(); + return cfg.config.model?.thinkingLevel; + }) + .toBe("high"); + // Reload only after the body is persisted via debounce: both the body and the two selections should restore from the cache. const draftKey = `penguin.chatDraft.${userId}.${projectId}`; await expect @@ -105,6 +125,8 @@ test("draft: pick model/approval -> reload restores them -> send creates the ses .toBe(true); await expect(page.getByRole("button", { name: "选择模型" })).toContainText("claude-4-8-mini"); await expect(page.getByRole("button", { name: "审批模式" })).toContainText("放行只读"); + // The thinking level is NOT draft state: it restores from the Agent config (written through above), not the cache. + await expect(page.getByRole("button", { name: "思考等级" })).toContainText("高"); // The @ target restores along with the draft; removing it falls back to a normal send (no delegation triggered). await expect(page.getByText("@agent_helper")).toBeVisible(); await page.getByRole("button", { name: "移除 @ 目标" }).click(); @@ -120,8 +142,24 @@ test("draft: pick model/approval -> reload restores them -> send creates the ses expect(first.session.provider).toBe("custom"); expect(first.session.approvalMode).toBe("read-only"); - // The cache clears as soon as sending succeeds. - await expect.poll(() => page.evaluate((k) => localStorage.getItem(k), draftKey)).toBeNull(); + // The written-through thinking level reached the session: its trace's session_meta records + // the level llmConfig was assembled with (per-session fixed), and the input area shows the + // read-only tag next to the locked model. + const replay = await ( + await page.request.get(`${BASE}/api/sessions/${firstSessionId}/messages`) + ).json(); + const meta = replay.messages.find((m) => m.type === "session_meta"); + expect(meta?.payload?.thinking_level).toBe("high"); + await expect(page.getByTitle("思考等级:高")).toBeVisible(); + + // On a successful send the cache clears — except the model selection, which carries over as + // the next conversation's default (switch-becomes-default, like the thinking level above). + await expect + .poll(() => page.evaluate((k) => localStorage.getItem(k), draftKey)) + .toContain("claude-4-8-mini"); + expect(await page.evaluate((k) => localStorage.getItem(k), draftKey)).not.toContain( + "Draft body must not be lost", + ); // —— Default grouping: the sidebar groups Sessions by Workspace — the session just created // used the auto temp directory, so it lands in the merged "临时工作区" group. —— @@ -185,8 +223,9 @@ test("draft: pick model/approval -> reload restores them -> send creates the ses expect(secondSessionId).not.toBe(firstSessionId); const second = await (await page.request.get(`${BASE}/api/sessions/${secondSessionId}`)).json(); expect(second.session.agentId).toBe("agent_helper"); - // No selection was changed: the model falls back to the project default, and approval mode falls back to allow-all (the previous draft was cleared, so read-only doesn't linger). - expect(second.session.modelId).toBe("claude-4-8"); + // The previously picked model carries over as the new default (switch-becomes-default); + // approval mode falls back to allow-all (the rest of the draft was cleared, so read-only doesn't linger). + expect(second.session.modelId).toBe("claude-4-8-mini"); expect(second.session.provider).toBe("custom"); expect(second.session.approvalMode).toBe("allow-all"); diff --git a/packages/web/src/features/chat/chat-input.tsx b/packages/web/src/features/chat/chat-input.tsx index f43760b..adb8ff6 100644 --- a/packages/web/src/features/chat/chat-input.tsx +++ b/packages/web/src/features/chat/chat-input.tsx @@ -11,6 +11,10 @@ * configured-key-first list with a bottom "show all" row — see ModelSelect) — once * the Session is created the model is locked, and the same spot switches to a read-only * logo + name display; + * Draft state also renders a thinking-level picker left of the model selector (backed by the + * Agent settings: picking a level writes through to the Agent config and applies to the session + * created on first send); in session state the level is fixed (llmConfig is assembled once per + * session), shown as a read-only tag from session_meta; * `/` opens the slash command menu (`/compact` compresses context, replacing the button; each * installed skill gets its own entry; pressing Enter on `/` toggles that skill's * selection without sending). Matching is positional like `@`: a slash opens the menu from any @@ -60,6 +64,7 @@ import { ProviderLogo } from "../../components/ui/provider-logo"; import { hasConfiguredKey, sameModelRef, visibleChatModels } from "../models/model-grouping"; import { filterAgents, matchMention, splitLeadingMention } from "./agent-mentions"; import { matchSlash, removeSlashToken } from "./slash-token"; +import { THINKING_LEVELS, thinkingLevelLabel } from "./thinking-level"; import { BOOK_ICON, buildSkillsMessage, @@ -379,6 +384,95 @@ function ModelSelect({ ); } +/** Spark glyph for the thinking-level picker (24x24 line path, consistent with the toolbar icon set). */ +const SPARK_ICON = "M12 3l1.9 5.1L19 10l-5.1 1.9L12 17l-1.9-5.1L5 10l5.1-1.9L12 3z"; + +/** + * Conversation-time thinking-level picker (draft state only, docked left of the model + * selector): shows the **selected Agent's** current `model.thinking_level` and writes a picked + * level straight through to the Agent settings — llmConfig is assembled once per session, so + * the level applies to the session created on first send and becomes the Agent's new default + * (switch-becomes-default). Per review: a title bar names the control, and the menu lists + * exactly the five levels with short names only (no descriptions, no "default" row) — an + * Agent without an explicit override shows an em dash until a level is picked. + */ +function ThinkingLevelSelect({ + value, + onChange, + disabled, +}: { + /** Current level ("" = no override yet); null = the Agent config is still loading. */ + value: string | null; + onChange: (level: string) => void; + disabled: boolean; +}) { + const [open, setOpen] = useState(false); + const label = + value === null ? "…" : (thinkingLevelLabel(S.chat.thinkingLevelNames, value) ?? "—"); + return ( + setOpen(!open)} + className="flex h-8 max-w-36 shrink-0 items-center gap-1.5 rounded-md px-2 text-xs text-gray-500 transition-colors duration-150 hover:bg-gray-100 hover:text-gray-800 disabled:cursor-not-allowed disabled:opacity-50 dark:text-gray-400 dark:hover:bg-gray-800 dark:hover:text-gray-200" + > + + {/* When the card is narrower than @md, only the icon remains (title shows the full state). */} + {label} + + + + + } + > + {/* Title bar: names the control (the rows themselves are just the five short names). */} +
+ {S.chat.thinkingLevel} +
+ {THINKING_LEVELS.map((level) => ( + + ))} +
+ ); +} + /** * Multi-select skills dropdown (bottom toolbar, after approval mode): styled like the model * selector — button = book icon + "Skills" label + selected-count badge (no badge at 0; when the @@ -588,6 +682,9 @@ export function ChatInput({ models, onChangeModel, defaultModel, + thinkingLevel, + onChangeThinkingLevel, + sessionThinkingLevel, contextWindow, contextNow, contextStale = false, @@ -629,6 +726,24 @@ export function ChatInput({ onChangeModel?: (ref: ModelRefDto) => void; /** Project default model (marked "default" on the selector's candidate item). */ defaultModel?: ModelRefDto; + /** + * Draft state: the selected Agent's current thinking level ("" = no override / provider + * default; null while the Agent config is loading — the picker renders disabled). Supplied + * together with onChangeThinkingLevel; without the callback the picker isn't rendered. + */ + thinkingLevel?: string | null; + /** + * Draft state: writes the picked level straight through to the Agent settings (the parent + * persists it via the agent-config API; the session created on first send picks it up and it + * becomes the Agent's new default). + */ + onChangeThinkingLevel?: (level: string) => void; + /** + * Session state: the session's fixed thinking level (captured from session_meta on replay; + * llmConfig is assembled once per session, so it cannot change mid-session). Rendered as a + * read-only tag next to the locked model; null/undefined = unknown (nothing shown). + */ + sessionThinkingLevel?: string | null; /** Model's context window (from models config; when not configured, the ring's cap falls back to 128000 via resolveContextWindow). */ contextWindow?: number; /** Current context usage (total of the most recent main-session Request). */ @@ -1265,6 +1380,31 @@ export function ChatInput({ {...(contextWindow !== undefined ? { window: contextWindow } : {})} /> )} + {/* Draft state: conversation-time thinking level (backed by Agent settings), docked left of the model selector. */} + {models && onChangeModel && onChangeThinkingLevel && ( + + )} + {/* Session state: the session's fixed thinking level (from session_meta), read-only next + to the locked model; hidden when it isn't one of the five levels (e.g. "default"). */} + {!onChangeModel && + (() => { + const label = thinkingLevelLabel(S.chat.thinkingLevelNames, sessionThinkingLevel); + return ( + label && ( + + + {label} + + ) + ); + })()} {/* Left of the send button: model selector in draft state; once the Session is created the model is locked, shown read-only (still with the provider logo). */} {models && onChangeModel ? ( (null); + useEffect(() => { + setThinkingLevel(null); + if (!agentId) return; + let cancelled = false; + api + .getAgentConfig(projectId, agentId) + .then((res) => { + if (!cancelled) setThinkingLevel(res.config.model?.thinkingLevel ?? ""); + }) + .catch(() => undefined); + return () => { + cancelled = true; + }; + }, [projectId, agentId]); + /** Live mirror for the rollback value (a stale closure would roll back to an outdated level). */ + const thinkingRef = useRef(null); + thinkingRef.current = thinkingLevel; + const onChangeThinkingLevel = useCallback( + (level: string) => { + // "" (no override) is not persistable through the config API — the picker disables that row. + if (!agentId || !level) return; + const rollback = thinkingRef.current; + setThinkingLevel(level); // Optimistic: the picker reflects the choice immediately. + api + .putAgentConfig(projectId, agentId, { + config: { model: { thinkingLevel: level as AgentModelConfigDto["thinkingLevel"] } }, + }) + .catch((e: unknown) => { + setThinkingLevel(rollback); + toastError(e instanceof ApiError ? e.message : S.common.unknownError); + }); + }, + [projectId, agentId], + ); + // Skills installed on the currently selected Agent (candidates for the input // area's skills dropdown): switching Agents first clears the list (which also // clears the selection in the input area), then refetches; a fetch failure is @@ -302,13 +348,21 @@ export function DraftView({ [], ); - /** Discard the draft after a successful send: first cancels the pending save timer, otherwise it would write the just-cleared draft back. */ + /** + * Discard the draft after a successful send: first cancels the pending save timer, otherwise + * it would write the just-cleared draft back. The **model selection carries over** as the + * next conversation's default (review: switching the model, like switching the thinking + * level, makes the switched-to value the new default — the level persists on the Agent + * config, the model here in the per-user draft cache); everything else clears. + */ const discardDraft = useCallback(() => { cancelPendingSave(); // Clear the preselected skills too: any subsequent write (e.g. the unmount flush) must not resurrect a selection that's already been sent. skillsRef.current = []; - if (userId) clearDraft(draftKey(userId, projectId)); - }, [cancelPendingSave, userId, projectId]); + if (!userId) return; + if (modelRef) saveDraft(draftKey(userId, projectId), { modelRef }); + else clearDraft(draftKey(userId, projectId)); + }, [cancelPendingSave, userId, projectId, modelRef]); const selectAgent = (a: AgentSummary) => { setAgentId(a.agentId); @@ -466,6 +520,8 @@ export function DraftView({ modelRef={modelRef} models={models?.models ?? []} onChangeModel={setModelRef} + thinkingLevel={thinkingLevel} + onChangeThinkingLevel={onChangeThinkingLevel} {...(models?.defaultModel !== undefined ? { defaultModel: models.defaultModel } : {})} {...(contextWindow !== undefined ? { contextWindow } : {})} contextNow={0} diff --git a/packages/web/src/features/chat/thinking-level.ts b/packages/web/src/features/chat/thinking-level.ts new file mode 100644 index 0000000..d421ffd --- /dev/null +++ b/packages/web/src/features/chat/thinking-level.ts @@ -0,0 +1,28 @@ +/** + * Pure logic for the conversation-time thinking-level picker (chat draft view). + * + * The picker is backed by the **Agent settings** (`system_config.model.thinking_level`): + * it shows the selected Agent's current level and writes a picked level straight through + * to the Agent config, so the session created on first send — which reads systemConfig + * fresh — runs with it, and it becomes the Agent's new default (switch-becomes-default). + * Per review: the menu lists exactly the five levels with short names only (no + * descriptions, no "default" row) under a title bar naming the control. + */ + +/** The five levels, in menu order (mirrors core's ThinkingLevelName). */ +export const THINKING_LEVELS = ["none", "low", "medium", "high", "xhigh"] as const; + +/** + * Short display label for a level from the localized name table (S.chat.thinkingLevelNames). + * Returns null for anything outside the five levels — including "" (an Agent without an + * explicit override) and session_meta's "default" — so callers can render a placeholder on + * the trigger and hide the session read-only tag instead of showing a raw internal value. + */ +export function thinkingLevelLabel( + names: Readonly>, + level: string | null | undefined, +): string | null { + return level && (THINKING_LEVELS as readonly string[]).includes(level) + ? (names[level] ?? level) + : null; +} diff --git a/packages/web/src/features/models/catalog-sync.ts b/packages/web/src/features/models/catalog-sync.ts index 7de6f6e..5d9fe33 100644 --- a/packages/web/src/features/models/catalog-sync.ts +++ b/packages/web/src/features/models/catalog-sync.ts @@ -32,6 +32,9 @@ function presetToRow(p: PresetEntry): RowState { modelId: p.model_id, original: null, ...presetFields(p), + // The output cap is user-owned, not catalog-owned (deliberately outside presetFields, + // so a sync never clobbers it on existing rows): fresh rows inherit the Agent setting. + maxTokens: "", originalBaseUrl: "", apiKeyInput: "", clearApiKey: false, diff --git a/packages/web/src/features/models/models-page.tsx b/packages/web/src/features/models/models-page.tsx index a3aeb37..dbd905b 100644 --- a/packages/web/src/features/models/models-page.tsx +++ b/packages/web/src/features/models/models-page.tsx @@ -177,6 +177,8 @@ export interface RowState { /** Environment variable name used as fallback when api_key is empty (given by the server based on catalog/protocol). */ envKey?: string; contextWindow: string; + /** Per-model max output tokens ("" = inherit the Agent setting): caps output per request; user-only, never preset by the catalog. */ + maxTokens: string; /** AgentHub client protocol: defaults for preset models (auto-routed), "openai" for new custom models; kept as-is, not user-editable. */ clientType: string; cacheRead: string; @@ -237,7 +239,10 @@ function FieldError({ text }: { text: string }) { /** Fields in the config dialog that can be highlighted red on error (keys match RowState field names, so they can be cleared per edit action). */ type FieldErrors = Partial< - Record<"modelId" | "baseUrl" | "contextWindow" | "cacheRead" | "cacheWrite" | "output", string> + Record< + "modelId" | "baseUrl" | "contextWindow" | "maxTokens" | "cacheRead" | "cacheWrite" | "output", + string + > >; /** @@ -270,6 +275,7 @@ export function toRow(m: ModelsResponse["models"][number]): RowState { original: { provider: m.provider, modelId: m.modelId }, vision: m.vision !== false, contextWindow: m.contextWindow !== undefined ? String(m.contextWindow) : "", + maxTokens: m.maxTokens !== undefined ? String(m.maxTokens) : "", clientType: m.clientType ?? "", cacheRead: m.pricing ? String(m.pricing.cacheRead) : "", cacheWrite: m.pricing ? String(m.pricing.cacheWrite) : "", @@ -300,6 +306,9 @@ function rowToEntry(row: RowState): ModelUpdateEntry { if (row.clientType.trim()) entry.clientType = row.clientType.trim(); // Supported by default: submit false only when explicitly marked "unsupported" (preset vision models and checked custom models aren't persisted). if (!row.vision) entry.vision = false; + // Output cap ("" = inherit the Agent setting): submitted only when filled; omitting clears the stored annotation. + const mt = Number(row.maxTokens.trim()); + if (row.maxTokens.trim() && Number.isFinite(mt) && mt > 0) entry.maxTokens = mt; const cr = Number(row.cacheRead.trim()); const cwr = Number(row.cacheWrite.trim()); const out = Number(row.output.trim()); @@ -1047,6 +1056,7 @@ function ModelDialog({ original: null, vision: true, contextWindow: "", + maxTokens: "", clientType: vendorAdd ? "" : "openai", cacheRead: "", cacheWrite: "", @@ -1161,6 +1171,14 @@ function ModelDialog({ if (contextWindow && !Number.isFinite(Number(contextWindow))) { errs.contextWindow = S.models.contextWindowInvalid; } + // Output cap: digits-only input can still hold "0"/pasted junk; the server requires a positive integer. + const maxTokensInput = form.maxTokens.trim(); + if ( + maxTokensInput && + !(Number.isInteger(Number(maxTokensInput)) && Number(maxTokensInput) > 0) + ) { + errs.maxTokens = S.models.maxTokensInvalid; + } if (Object.keys(errs).length > 0) { setFieldErrors(errs); @@ -1517,7 +1535,33 @@ function ModelDialog({ {fieldErrors.contextWindow && } - {/* 4) Pricing: three fields side by side; currency and unit (/M tok) both + {/* 4) Max output tokens: per-model cap on the request's output — when set it wins + over the Agent's system_config value; empty inherits it. Lets a small-context + local model stay under its window (the per-Agent default may not fit). */} + + + {/* 5) Pricing: three fields side by side; currency and unit (/M tok) both shown inside the input, no need to repeat in the title. */}

@@ -1557,7 +1601,7 @@ function ModelDialog({

- {/* 5) Identity: model id (renamable) + display name and group (side by side) */} + {/* 6) Identity: model id (renamable) + display name and group (side by side) */} {!isNew && identityFields} {/* Legacy entries carrying a non-openai client_type (historical config): read-only display. */} {!isNew && !preset && form.clientType && form.clientType !== "openai" && ( diff --git a/packages/web/src/lib/omni/stream-model.ts b/packages/web/src/lib/omni/stream-model.ts index 6080ccc..7b7c57e 100644 --- a/packages/web/src/lib/omni/stream-model.ts +++ b/packages/web/src/lib/omni/stream-model.ts @@ -241,6 +241,13 @@ export interface StreamModel { items: ChatItem[]; /** A nested sub-session model (produces no stats row; its stats count toward the parent). */ nested: boolean; + /** + * The session's thinking level, captured from the main session's `session_meta` on history + * replay ("default" when the Agent config leaves it unset; null until a session_meta has been + * seen). llmConfig is assembled once per session, so this is fixed for the session's lifetime — + * shown read-only in the input area next to the locked model. + */ + thinkingLevel: string | null; stats: TaskStatsTracker; /** The currently open text/thinking fragment (opened by start, closed by stop). */ openText: AssistantTextItem | null; @@ -318,6 +325,7 @@ function newModel(nested: boolean, localDecisions: Set): StreamModel { return { items: [], nested, + thinkingLevel: null, stats: createTaskStatsTracker(), openText: null, openThinking: null, @@ -392,7 +400,13 @@ export function pushMessage( advanceLastTs(model, msg.timestamp); return; } - // session_meta (main session): not rendered. + // session_meta (main session): not rendered as an item, but its thinking level is captured + // for the input area's read-only display (fixed per session: llmConfig is assembled once at + // session creation, so a mid-conversation Agent-config change doesn't affect this session). + if (msg.type === "session_meta") { + const level = (msg.payload as { thinking_level?: unknown }).thinking_level; + if (typeof level === "string" && level) model.thinkingLevel = level; + } } /** ISO timestamp → milliseconds (returns undefined if invalid). */ diff --git a/packages/web/src/lib/strings-en.ts b/packages/web/src/lib/strings-en.ts index 579ae1c..e60acb4 100644 --- a/packages/web/src/lib/strings-en.ts +++ b/packages/web/src/lib/strings-en.ts @@ -312,6 +312,10 @@ export const en: Strings = { contextWindow: "Context window", contextWindowUnit: "tokens", contextWindowHint: "Leave empty if unknown", + maxTokens: "Max output tokens", + maxTokensHint: "Leave empty to inherit the agent setting", + maxTokensCapHint: "Caps output tokens per request; lower it for small-context local models", + maxTokensInvalid: "Must be a positive integer", clientTypeLocked: (t: string): string => `Protocol: ${t} (kept as configured; not editable)`, vision: "Supports image input (vision / multimodal)", visionHint: @@ -480,6 +484,14 @@ export const en: Strings = { newSessionMenu: "New chat", chooseAgent: "Choose agent", chooseModel: "Choose model", + thinkingLevel: "Thinking level", + thinkingLevelNames: { + none: "None", + low: "Low", + medium: "Medium", + high: "High", + xhigh: "Extreme High", + }, workspaceUseThis: "Use this dir", workspaceUp: "Parent dir", workspaceNoSubdirs: "No subdirectories", diff --git a/packages/web/src/lib/strings.ts b/packages/web/src/lib/strings.ts index 31eccad..2bbf191 100644 --- a/packages/web/src/lib/strings.ts +++ b/packages/web/src/lib/strings.ts @@ -287,6 +287,10 @@ export const zh = { contextWindow: "上下文窗口", contextWindowUnit: "tokens", contextWindowHint: "留空表示未知", + maxTokens: "最大输出长度(Token)", + maxTokensHint: "留空沿用 Agent 设置", + maxTokensCapHint: "限制单次请求的输出 Token 数;小上下文的本地模型建议调低", + maxTokensInvalid: "必须为正整数", clientTypeLocked: (t: string): string => `协议:${t}(沿用原配置,不可修改)`, vision: "支持图片输入(视觉/多模态)", visionHint: @@ -458,6 +462,15 @@ export const zh = { newSessionMenu: "新建对话", chooseAgent: "选择 Agent", chooseModel: "选择模型", + thinkingLevel: "思考等级", + /** 会话前拾取器的档位短名(评审要求:只写短名、不带说明、无“缺省”项)。 */ + thinkingLevelNames: { + none: "无", + low: "低", + medium: "中", + high: "高", + xhigh: "极高", + } as Readonly>, workspaceUseThis: "使用此目录", workspaceUp: "上级目录", workspaceNoSubdirs: "无子目录", diff --git a/packages/web/test/catalog-sync.test.ts b/packages/web/test/catalog-sync.test.ts index e066d11..739d327 100644 --- a/packages/web/test/catalog-sync.test.ts +++ b/packages/web/test/catalog-sync.test.ts @@ -16,6 +16,7 @@ function makeRow(partial: Partial & Pick { expect(rows[0]).toBe(upToDate); }); + it("preserves a user-set max output tokens through a preset sync (user-owned, not catalog-owned)", () => { + const local = makeRow({ + provider: "deepseek", + modelId: "deepseek-v4-pro", + maxTokens: "4096", // user annotation + contextWindow: "500000", // stale -> the row does get updated by the sync + }); + const { rows, updated } = syncRowsWithCatalog([local], PRESET); + expect(updated).toBe(1); + const row = rows[0]!; + expect(row.contextWindow).toBe("1000000"); // catalog-owned field reset + expect(row.maxTokens).toBe("4096"); // user field survives the {...row, ...fields} merge + // Fresh catalog rows default to inherit (no preset output cap exists). + expect(rows.find((r) => r.modelId === "glm-5.2")!.maxTokens).toBe(""); + }); + it("keeps locally added models (including user-defined groups) verbatim and in place", () => { const mine = makeRow({ provider: "my-gateway", modelId: "my-model", baseUrl: "http://x" }); const { rows, added } = syncRowsWithCatalog([mine], PRESET); diff --git a/packages/web/test/models-input.test.ts b/packages/web/test/models-input.test.ts index d72de0c..cb3640f 100644 --- a/packages/web/test/models-input.test.ts +++ b/packages/web/test/models-input.test.ts @@ -67,6 +67,18 @@ describe("toRow (DTO → row edit state)", () => { expect(row.provider).toBe("openrouter"); expect(row.modelId).toBe("xiaomi/mimo-v2.5"); }); + + it("carries the per-model max output tokens through; absent = '' (inherit the Agent setting)", () => { + const capped = toRow({ + provider: "custom", + modelId: "local-qwen", + maxTokens: 8000, + isDefault: false, + }); + expect(capped.maxTokens).toBe("8000"); + const plain = toRow({ provider: "custom", modelId: "local-qwen", isDefault: false }); + expect(plain.maxTokens).toBe(""); + }); }); describe("nextPointers (where the default/vision-agent model pointers land after save; always paired)", () => { diff --git a/packages/web/test/stream-model.test.ts b/packages/web/test/stream-model.test.ts index aa1ebbf..ba6d668 100644 --- a/packages/web/test/stream-model.test.ts +++ b/packages/web/test/stream-model.test.ts @@ -279,6 +279,34 @@ describe("approvals and events", () => { expect(items(m)).toHaveLength(0); }); + it("captures the session's thinking level from the main session_meta (read-only input-area tag)", () => { + const m = createStreamModel(); + expect(m.thinkingLevel).toBeNull(); + // The shared helper's meta carries thinking_level "default" (Agent config leaves it unset). + pushMessage(m, meta("session-x")); + expect(m.thinkingLevel).toBe("default"); + + const m2 = createStreamModel(); + pushMessage( + m2, + sessionMeta({ + session_id: "session-y", + model_id: "m", + provider: "custom", + model_context_window: 200000, + system_prompt: "", + tools: [], + thinking_level: "medium", + agent_state: "/a", + workspace: "/w", + }), + ); + expect(m2.thinkingLevel).toBe("medium"); + // An origin-tagged (sub-session) session_meta routes to the nested model and must not clobber the main session's level. + pushMessage(m2, withOrigin(meta("child"), "child")); + expect(m2.thinkingLevel).toBe("medium"); + }); + it("request_end final state timeout/malformed produces a retry notice item (with attempt number); request_begin marks it as resent", () => { const m = createStreamModel(); pushMessage(m, requestBegin()); diff --git a/packages/web/test/thinking-level.test.ts b/packages/web/test/thinking-level.test.ts new file mode 100644 index 0000000..fdce904 --- /dev/null +++ b/packages/web/test/thinking-level.test.ts @@ -0,0 +1,41 @@ +/** + * thinking-level.ts unit tests: the conversation-time picker's short-name lookup — the five + * levels map to their localized names, while "" (no override yet) and session_meta's + * "default" resolve to null (trigger shows a placeholder; the session tag hides). + */ +import { describe, expect, it } from "vitest"; +import { THINKING_LEVELS, thinkingLevelLabel } from "../src/features/chat/thinking-level"; + +/** Mirrors the shape of S.chat.thinkingLevelNames. */ +const NAMES: Readonly> = { + none: "None", + low: "Low", + medium: "Medium", + high: "High", + xhigh: "Extreme High", +}; + +describe("thinkingLevelLabel", () => { + it("maps each of the five levels to its localized short name (menu order preserved)", () => { + expect(THINKING_LEVELS).toEqual(["none", "low", "medium", "high", "xhigh"]); + expect(THINKING_LEVELS.map((l) => thinkingLevelLabel(NAMES, l))).toEqual([ + "None", + "Low", + "Medium", + "High", + "Extreme High", + ]); + }); + + it("returns null for non-levels: '' (no override yet), session_meta's 'default', unknown, null", () => { + expect(thinkingLevelLabel(NAMES, "")).toBeNull(); + expect(thinkingLevelLabel(NAMES, "default")).toBeNull(); + expect(thinkingLevelLabel(NAMES, "ultra")).toBeNull(); + expect(thinkingLevelLabel(NAMES, null)).toBeNull(); + expect(thinkingLevelLabel(NAMES, undefined)).toBeNull(); + }); + + it("falls back to the raw value if the name table misses a level (defensive)", () => { + expect(thinkingLevelLabel({}, "medium")).toBe("medium"); + }); +});