feat: per-model max output tokens and a conversation-time thinking level backed by agent settings (#28)
Co-authored-by: Alice <alice@prismshadow.com> Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -11,6 +11,10 @@
|
||||
* configured-key-first list with a bottom "show all" row — see ModelSelect) — once
|
||||
* the Session is created the model is locked, and the same spot switches to a read-only
|
||||
* logo + name display;
|
||||
* Draft state also renders a thinking-level picker left of the model selector (backed by the
|
||||
* Agent settings: picking a level writes through to the Agent config and applies to the session
|
||||
* created on first send); in session state the level is fixed (llmConfig is assembled once per
|
||||
* session), shown as a read-only tag from session_meta;
|
||||
* `/` opens the slash command menu (`/compact` compresses context, replacing the button; each
|
||||
* installed skill gets its own entry; pressing Enter on `/<skill_name>` toggles that skill's
|
||||
* selection without sending). Matching is positional like `@`: a slash opens the menu from any
|
||||
@@ -60,6 +64,7 @@ import { ProviderLogo } from "../../components/ui/provider-logo";
|
||||
import { hasConfiguredKey, sameModelRef, visibleChatModels } from "../models/model-grouping";
|
||||
import { filterAgents, matchMention, splitLeadingMention } from "./agent-mentions";
|
||||
import { matchSlash, removeSlashToken } from "./slash-token";
|
||||
import { THINKING_LEVELS, thinkingLevelLabel } from "./thinking-level";
|
||||
import {
|
||||
BOOK_ICON,
|
||||
buildSkillsMessage,
|
||||
@@ -379,6 +384,95 @@ function ModelSelect({
|
||||
);
|
||||
}
|
||||
|
||||
/** Spark glyph for the thinking-level picker (24x24 line path, consistent with the toolbar icon set). */
|
||||
const SPARK_ICON = "M12 3l1.9 5.1L19 10l-5.1 1.9L12 17l-1.9-5.1L5 10l5.1-1.9L12 3z";
|
||||
|
||||
/**
|
||||
* Conversation-time thinking-level picker (draft state only, docked left of the model
|
||||
* selector): shows the **selected Agent's** current `model.thinking_level` and writes a picked
|
||||
* level straight through to the Agent settings — llmConfig is assembled once per session, so
|
||||
* the level applies to the session created on first send and becomes the Agent's new default
|
||||
* (switch-becomes-default). Per review: a title bar names the control, and the menu lists
|
||||
* exactly the five levels with short names only (no descriptions, no "default" row) — an
|
||||
* Agent without an explicit override shows an em dash until a level is picked.
|
||||
*/
|
||||
function ThinkingLevelSelect({
|
||||
value,
|
||||
onChange,
|
||||
disabled,
|
||||
}: {
|
||||
/** Current level ("" = no override yet); null = the Agent config is still loading. */
|
||||
value: string | null;
|
||||
onChange: (level: string) => void;
|
||||
disabled: boolean;
|
||||
}) {
|
||||
const [open, setOpen] = useState(false);
|
||||
const label =
|
||||
value === null ? "…" : (thinkingLevelLabel(S.chat.thinkingLevelNames, value) ?? "—");
|
||||
return (
|
||||
<Dropdown
|
||||
open={open}
|
||||
setOpen={setOpen}
|
||||
menuClass="right-0 top-full mt-1 w-36 origin-top-right"
|
||||
button={
|
||||
<button
|
||||
type="button"
|
||||
title={`${S.chat.thinkingLevel}:${label}`}
|
||||
aria-label={S.chat.thinkingLevel}
|
||||
disabled={disabled || value === null}
|
||||
onClick={() => setOpen(!open)}
|
||||
className="flex h-8 max-w-36 shrink-0 items-center gap-1.5 rounded-md px-2 text-xs text-gray-500 transition-colors duration-150 hover:bg-gray-100 hover:text-gray-800 disabled:cursor-not-allowed disabled:opacity-50 dark:text-gray-400 dark:hover:bg-gray-800 dark:hover:text-gray-200"
|
||||
>
|
||||
<GlyphIcon d={SPARK_ICON} size={14} className="shrink-0" />
|
||||
{/* When the card is narrower than @md, only the icon remains (title shows the full state). */}
|
||||
<span className="hidden min-w-0 truncate @md:block">{label}</span>
|
||||
<svg
|
||||
width="10"
|
||||
height="10"
|
||||
viewBox="0 0 12 12"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
className="shrink-0"
|
||||
aria-hidden
|
||||
>
|
||||
<path
|
||||
d="M3 4.5l3 3 3-3"
|
||||
strokeWidth="1.5"
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
/>
|
||||
</svg>
|
||||
</button>
|
||||
}
|
||||
>
|
||||
{/* Title bar: names the control (the rows themselves are just the five short names). */}
|
||||
<div className="border-b border-gray-100 px-3 pb-1.5 pt-0.5 text-xs font-semibold text-gray-500 dark:border-gray-800 dark:text-gray-400">
|
||||
{S.chat.thinkingLevel}
|
||||
</div>
|
||||
{THINKING_LEVELS.map((level) => (
|
||||
<button
|
||||
key={level}
|
||||
type="button"
|
||||
onClick={() => {
|
||||
onChange(level);
|
||||
setOpen(false);
|
||||
}}
|
||||
className={`flex w-full items-center gap-2 px-3 py-1.5 text-left text-xs transition-colors duration-150 hover:bg-gray-100 dark:hover:bg-gray-800 ${
|
||||
level === value
|
||||
? "font-medium text-gray-900 dark:text-gray-100"
|
||||
: "text-gray-600 dark:text-gray-400"
|
||||
}`}
|
||||
>
|
||||
<span className="min-w-0 flex-1 truncate">
|
||||
{S.chat.thinkingLevelNames[level] ?? level}
|
||||
</span>
|
||||
<span className="w-3 shrink-0 text-center">{level === value ? "✓" : ""}</span>
|
||||
</button>
|
||||
))}
|
||||
</Dropdown>
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Multi-select skills dropdown (bottom toolbar, after approval mode): styled like the model
|
||||
* selector — button = book icon + "Skills" label + selected-count badge (no badge at 0; when the
|
||||
@@ -588,6 +682,9 @@ export function ChatInput({
|
||||
models,
|
||||
onChangeModel,
|
||||
defaultModel,
|
||||
thinkingLevel,
|
||||
onChangeThinkingLevel,
|
||||
sessionThinkingLevel,
|
||||
contextWindow,
|
||||
contextNow,
|
||||
contextStale = false,
|
||||
@@ -629,6 +726,24 @@ export function ChatInput({
|
||||
onChangeModel?: (ref: ModelRefDto) => void;
|
||||
/** Project default model (marked "default" on the selector's candidate item). */
|
||||
defaultModel?: ModelRefDto;
|
||||
/**
|
||||
* Draft state: the selected Agent's current thinking level ("" = no override / provider
|
||||
* default; null while the Agent config is loading — the picker renders disabled). Supplied
|
||||
* together with onChangeThinkingLevel; without the callback the picker isn't rendered.
|
||||
*/
|
||||
thinkingLevel?: string | null;
|
||||
/**
|
||||
* Draft state: writes the picked level straight through to the Agent settings (the parent
|
||||
* persists it via the agent-config API; the session created on first send picks it up and it
|
||||
* becomes the Agent's new default).
|
||||
*/
|
||||
onChangeThinkingLevel?: (level: string) => void;
|
||||
/**
|
||||
* Session state: the session's fixed thinking level (captured from session_meta on replay;
|
||||
* llmConfig is assembled once per session, so it cannot change mid-session). Rendered as a
|
||||
* read-only tag next to the locked model; null/undefined = unknown (nothing shown).
|
||||
*/
|
||||
sessionThinkingLevel?: string | null;
|
||||
/** Model's context window (from models config; when not configured, the ring's cap falls back to 128000 via resolveContextWindow). */
|
||||
contextWindow?: number;
|
||||
/** Current context usage (total of the most recent main-session Request). */
|
||||
@@ -1265,6 +1380,31 @@ export function ChatInput({
|
||||
{...(contextWindow !== undefined ? { window: contextWindow } : {})}
|
||||
/>
|
||||
)}
|
||||
{/* Draft state: conversation-time thinking level (backed by Agent settings), docked left of the model selector. */}
|
||||
{models && onChangeModel && onChangeThinkingLevel && (
|
||||
<ThinkingLevelSelect
|
||||
value={thinkingLevel ?? null}
|
||||
onChange={onChangeThinkingLevel}
|
||||
disabled={busy}
|
||||
/>
|
||||
)}
|
||||
{/* Session state: the session's fixed thinking level (from session_meta), read-only next
|
||||
to the locked model; hidden when it isn't one of the five levels (e.g. "default"). */}
|
||||
{!onChangeModel &&
|
||||
(() => {
|
||||
const label = thinkingLevelLabel(S.chat.thinkingLevelNames, sessionThinkingLevel);
|
||||
return (
|
||||
label && (
|
||||
<span
|
||||
title={`${S.chat.thinkingLevel}:${label}`}
|
||||
className="hidden h-8 shrink-0 items-center gap-1 px-1 text-xs text-gray-400 @md:flex dark:text-gray-500"
|
||||
>
|
||||
<GlyphIcon d={SPARK_ICON} size={13} className="shrink-0" />
|
||||
{label}
|
||||
</span>
|
||||
)
|
||||
);
|
||||
})()}
|
||||
{/* Left of the send button: model selector in draft state; once the Session is created the model is locked, shown read-only (still with the provider logo). */}
|
||||
{models && onChangeModel ? (
|
||||
<ModelSelect
|
||||
|
||||
@@ -494,6 +494,7 @@ export function ChatPage() {
|
||||
onCompact={onCompact}
|
||||
modelRef={activeModelRef}
|
||||
{...(models !== null ? { models: models.models } : {})}
|
||||
sessionThinkingLevel={stream.model.thinkingLevel}
|
||||
{...(contextWindow !== undefined ? { contextWindow } : {})}
|
||||
contextNow={stream.model.stats.contextNow}
|
||||
contextStale={stream.model.stats.contextStale}
|
||||
|
||||
@@ -15,7 +15,9 @@
|
||||
* "user × Project", #68): the four selections are saved as soon as they change;
|
||||
* body text is keystroke-frequent and deferred/coalesced (if there's an unsaved
|
||||
* change before unmount, one final write is flushed) — closing and returning to
|
||||
* the page resumes where you left off; the cache is cleared on successful send.
|
||||
* the page resumes where you left off; on successful send the cache clears, except
|
||||
* the model selection, which carries over as the next conversation's default
|
||||
* (switch-becomes-default, mirroring the thinking level persisting on the Agent).
|
||||
* The sidebar group header "+" / menu "New conversation" explicitly specify an
|
||||
* Agent via route state (overriding the cached selection); the workspace-mode
|
||||
* group header "+" additionally carries a Workspace path pre-filling the
|
||||
@@ -25,6 +27,7 @@
|
||||
import { useCallback, useEffect, useRef, useState } from "react";
|
||||
import { useLocation, useNavigate } from "react-router";
|
||||
import type {
|
||||
AgentModelConfigDto,
|
||||
AgentSummary,
|
||||
ApprovalMode,
|
||||
DirListResponse,
|
||||
@@ -204,6 +207,49 @@ export function DraftView({
|
||||
);
|
||||
}, [models, modelRef]);
|
||||
|
||||
// —— Conversation-time thinking level (backed by the Agent settings) ——
|
||||
// Shows the selected Agent's current `model.thinking_level` ("" = no override); picking a
|
||||
// level immediately persists it via the agent-config API (the PUT carries only that key —
|
||||
// the server merges per-key into the YAML, so nothing else is clobbered). The session created
|
||||
// on first send reads systemConfig fresh, so it runs with the picked level, which also
|
||||
// becomes the Agent's new default. Refetched whenever the draft's Agent changes; while
|
||||
// loading (or after a failed fetch) the picker stays disabled (null).
|
||||
const [thinkingLevel, setThinkingLevel] = useState<string | null>(null);
|
||||
useEffect(() => {
|
||||
setThinkingLevel(null);
|
||||
if (!agentId) return;
|
||||
let cancelled = false;
|
||||
api
|
||||
.getAgentConfig(projectId, agentId)
|
||||
.then((res) => {
|
||||
if (!cancelled) setThinkingLevel(res.config.model?.thinkingLevel ?? "");
|
||||
})
|
||||
.catch(() => undefined);
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [projectId, agentId]);
|
||||
/** Live mirror for the rollback value (a stale closure would roll back to an outdated level). */
|
||||
const thinkingRef = useRef<string | null>(null);
|
||||
thinkingRef.current = thinkingLevel;
|
||||
const onChangeThinkingLevel = useCallback(
|
||||
(level: string) => {
|
||||
// "" (no override) is not persistable through the config API — the picker disables that row.
|
||||
if (!agentId || !level) return;
|
||||
const rollback = thinkingRef.current;
|
||||
setThinkingLevel(level); // Optimistic: the picker reflects the choice immediately.
|
||||
api
|
||||
.putAgentConfig(projectId, agentId, {
|
||||
config: { model: { thinkingLevel: level as AgentModelConfigDto["thinkingLevel"] } },
|
||||
})
|
||||
.catch((e: unknown) => {
|
||||
setThinkingLevel(rollback);
|
||||
toastError(e instanceof ApiError ? e.message : S.common.unknownError);
|
||||
});
|
||||
},
|
||||
[projectId, agentId],
|
||||
);
|
||||
|
||||
// Skills installed on the currently selected Agent (candidates for the input
|
||||
// area's skills dropdown): switching Agents first clears the list (which also
|
||||
// clears the selection in the input area), then refetches; a fetch failure is
|
||||
@@ -302,13 +348,21 @@ export function DraftView({
|
||||
[],
|
||||
);
|
||||
|
||||
/** Discard the draft after a successful send: first cancels the pending save timer, otherwise it would write the just-cleared draft back. */
|
||||
/**
|
||||
* Discard the draft after a successful send: first cancels the pending save timer, otherwise
|
||||
* it would write the just-cleared draft back. The **model selection carries over** as the
|
||||
* next conversation's default (review: switching the model, like switching the thinking
|
||||
* level, makes the switched-to value the new default — the level persists on the Agent
|
||||
* config, the model here in the per-user draft cache); everything else clears.
|
||||
*/
|
||||
const discardDraft = useCallback(() => {
|
||||
cancelPendingSave();
|
||||
// Clear the preselected skills too: any subsequent write (e.g. the unmount flush) must not resurrect a selection that's already been sent.
|
||||
skillsRef.current = [];
|
||||
if (userId) clearDraft(draftKey(userId, projectId));
|
||||
}, [cancelPendingSave, userId, projectId]);
|
||||
if (!userId) return;
|
||||
if (modelRef) saveDraft(draftKey(userId, projectId), { modelRef });
|
||||
else clearDraft(draftKey(userId, projectId));
|
||||
}, [cancelPendingSave, userId, projectId, modelRef]);
|
||||
|
||||
const selectAgent = (a: AgentSummary) => {
|
||||
setAgentId(a.agentId);
|
||||
@@ -466,6 +520,8 @@ export function DraftView({
|
||||
modelRef={modelRef}
|
||||
models={models?.models ?? []}
|
||||
onChangeModel={setModelRef}
|
||||
thinkingLevel={thinkingLevel}
|
||||
onChangeThinkingLevel={onChangeThinkingLevel}
|
||||
{...(models?.defaultModel !== undefined ? { defaultModel: models.defaultModel } : {})}
|
||||
{...(contextWindow !== undefined ? { contextWindow } : {})}
|
||||
contextNow={0}
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
/**
|
||||
* Pure logic for the conversation-time thinking-level picker (chat draft view).
|
||||
*
|
||||
* The picker is backed by the **Agent settings** (`system_config.model.thinking_level`):
|
||||
* it shows the selected Agent's current level and writes a picked level straight through
|
||||
* to the Agent config, so the session created on first send — which reads systemConfig
|
||||
* fresh — runs with it, and it becomes the Agent's new default (switch-becomes-default).
|
||||
* Per review: the menu lists exactly the five levels with short names only (no
|
||||
* descriptions, no "default" row) under a title bar naming the control.
|
||||
*/
|
||||
|
||||
/** The five levels, in menu order (mirrors core's ThinkingLevelName). */
|
||||
export const THINKING_LEVELS = ["none", "low", "medium", "high", "xhigh"] as const;
|
||||
|
||||
/**
|
||||
* Short display label for a level from the localized name table (S.chat.thinkingLevelNames).
|
||||
* Returns null for anything outside the five levels — including "" (an Agent without an
|
||||
* explicit override) and session_meta's "default" — so callers can render a placeholder on
|
||||
* the trigger and hide the session read-only tag instead of showing a raw internal value.
|
||||
*/
|
||||
export function thinkingLevelLabel(
|
||||
names: Readonly<Record<string, string>>,
|
||||
level: string | null | undefined,
|
||||
): string | null {
|
||||
return level && (THINKING_LEVELS as readonly string[]).includes(level)
|
||||
? (names[level] ?? level)
|
||||
: null;
|
||||
}
|
||||
@@ -32,6 +32,9 @@ function presetToRow(p: PresetEntry): RowState {
|
||||
modelId: p.model_id,
|
||||
original: null,
|
||||
...presetFields(p),
|
||||
// The output cap is user-owned, not catalog-owned (deliberately outside presetFields,
|
||||
// so a sync never clobbers it on existing rows): fresh rows inherit the Agent setting.
|
||||
maxTokens: "",
|
||||
originalBaseUrl: "",
|
||||
apiKeyInput: "",
|
||||
clearApiKey: false,
|
||||
|
||||
@@ -177,6 +177,8 @@ export interface RowState {
|
||||
/** Environment variable name used as fallback when api_key is empty (given by the server based on catalog/protocol). */
|
||||
envKey?: string;
|
||||
contextWindow: string;
|
||||
/** Per-model max output tokens ("" = inherit the Agent setting): caps output per request; user-only, never preset by the catalog. */
|
||||
maxTokens: string;
|
||||
/** AgentHub client protocol: defaults for preset models (auto-routed), "openai" for new custom models; kept as-is, not user-editable. */
|
||||
clientType: string;
|
||||
cacheRead: string;
|
||||
@@ -237,7 +239,10 @@ function FieldError({ text }: { text: string }) {
|
||||
|
||||
/** Fields in the config dialog that can be highlighted red on error (keys match RowState field names, so they can be cleared per edit action). */
|
||||
type FieldErrors = Partial<
|
||||
Record<"modelId" | "baseUrl" | "contextWindow" | "cacheRead" | "cacheWrite" | "output", string>
|
||||
Record<
|
||||
"modelId" | "baseUrl" | "contextWindow" | "maxTokens" | "cacheRead" | "cacheWrite" | "output",
|
||||
string
|
||||
>
|
||||
>;
|
||||
|
||||
/**
|
||||
@@ -270,6 +275,7 @@ export function toRow(m: ModelsResponse["models"][number]): RowState {
|
||||
original: { provider: m.provider, modelId: m.modelId },
|
||||
vision: m.vision !== false,
|
||||
contextWindow: m.contextWindow !== undefined ? String(m.contextWindow) : "",
|
||||
maxTokens: m.maxTokens !== undefined ? String(m.maxTokens) : "",
|
||||
clientType: m.clientType ?? "",
|
||||
cacheRead: m.pricing ? String(m.pricing.cacheRead) : "",
|
||||
cacheWrite: m.pricing ? String(m.pricing.cacheWrite) : "",
|
||||
@@ -300,6 +306,9 @@ function rowToEntry(row: RowState): ModelUpdateEntry {
|
||||
if (row.clientType.trim()) entry.clientType = row.clientType.trim();
|
||||
// Supported by default: submit false only when explicitly marked "unsupported" (preset vision models and checked custom models aren't persisted).
|
||||
if (!row.vision) entry.vision = false;
|
||||
// Output cap ("" = inherit the Agent setting): submitted only when filled; omitting clears the stored annotation.
|
||||
const mt = Number(row.maxTokens.trim());
|
||||
if (row.maxTokens.trim() && Number.isFinite(mt) && mt > 0) entry.maxTokens = mt;
|
||||
const cr = Number(row.cacheRead.trim());
|
||||
const cwr = Number(row.cacheWrite.trim());
|
||||
const out = Number(row.output.trim());
|
||||
@@ -1047,6 +1056,7 @@ function ModelDialog({
|
||||
original: null,
|
||||
vision: true,
|
||||
contextWindow: "",
|
||||
maxTokens: "",
|
||||
clientType: vendorAdd ? "" : "openai",
|
||||
cacheRead: "",
|
||||
cacheWrite: "",
|
||||
@@ -1161,6 +1171,14 @@ function ModelDialog({
|
||||
if (contextWindow && !Number.isFinite(Number(contextWindow))) {
|
||||
errs.contextWindow = S.models.contextWindowInvalid;
|
||||
}
|
||||
// Output cap: digits-only input can still hold "0"/pasted junk; the server requires a positive integer.
|
||||
const maxTokensInput = form.maxTokens.trim();
|
||||
if (
|
||||
maxTokensInput &&
|
||||
!(Number.isInteger(Number(maxTokensInput)) && Number(maxTokensInput) > 0)
|
||||
) {
|
||||
errs.maxTokens = S.models.maxTokensInvalid;
|
||||
}
|
||||
|
||||
if (Object.keys(errs).length > 0) {
|
||||
setFieldErrors(errs);
|
||||
@@ -1517,7 +1535,33 @@ function ModelDialog({
|
||||
{fieldErrors.contextWindow && <FieldError text={fieldErrors.contextWindow} />}
|
||||
</label>
|
||||
|
||||
{/* 4) Pricing: three fields side by side; currency and unit (/M tok) both
|
||||
{/* 4) Max output tokens: per-model cap on the request's output — when set it wins
|
||||
over the Agent's system_config value; empty inherits it. Lets a small-context
|
||||
local model stay under its window (the per-Agent default may not fit). */}
|
||||
<label className="block">
|
||||
<span className="mb-1 block text-xs font-semibold text-gray-600 dark:text-gray-400">
|
||||
{S.models.maxTokens}
|
||||
</span>
|
||||
<Input
|
||||
size="sm"
|
||||
value={form.maxTokens}
|
||||
inputMode="numeric"
|
||||
disabled={!canEdit}
|
||||
invalid={Boolean(fieldErrors.maxTokens)}
|
||||
onChange={(e) => set({ maxTokens: digitsOnly(e.target.value) })}
|
||||
className="font-mono"
|
||||
placeholder={S.models.maxTokensHint}
|
||||
/>
|
||||
{fieldErrors.maxTokens ? (
|
||||
<FieldError text={fieldErrors.maxTokens} />
|
||||
) : (
|
||||
<span className="mt-1 block text-xs text-gray-400 dark:text-gray-500">
|
||||
{S.models.maxTokensCapHint}
|
||||
</span>
|
||||
)}
|
||||
</label>
|
||||
|
||||
{/* 5) Pricing: three fields side by side; currency and unit (/M tok) both
|
||||
shown inside the input, no need to repeat in the title. */}
|
||||
<div>
|
||||
<p className="mb-1.5 text-xs font-semibold text-gray-600 dark:text-gray-400">
|
||||
@@ -1557,7 +1601,7 @@ function ModelDialog({
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* 5) Identity: model id (renamable) + display name and group (side by side) */}
|
||||
{/* 6) Identity: model id (renamable) + display name and group (side by side) */}
|
||||
{!isNew && identityFields}
|
||||
{/* Legacy entries carrying a non-openai client_type (historical config): read-only display. */}
|
||||
{!isNew && !preset && form.clientType && form.clientType !== "openai" && (
|
||||
|
||||
Reference in New Issue
Block a user