feat: per-model max output tokens and a conversation-time thinking level backed by agent settings (#28)

Co-authored-by: Alice <alice@prismshadow.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-07-22 22:22:41 +08:00
committed by GitHub
parent 99f391f379
commit cabbab1c16
36 changed files with 797 additions and 42 deletions
+9
View File
@@ -206,6 +206,13 @@ export interface ModelInfo {
* unset (= treated as supported).
*/
vision?: boolean;
/**
* Per-model max output tokens (TOML `max_tokens` annotation; user-only, never preset by the
* built-in catalog): when set it wins over the Agent's `system_config.model.max_tokens`;
* unset = inherit the Agent value. Lets a small-context model cap its output below the
* seeded per-Agent default (32000), which cannot fit into e.g. a 32k context window.
*/
maxTokens?: number;
pricing?: ModelPricingDto;
/** Environment variable name to fall back to when api_key is empty (e.g. ANTHROPIC_API_KEY); unset if no known fallback. */
envKey?: string;
@@ -241,6 +248,8 @@ export interface ModelUpdateEntry {
clientType?: string;
/** Whether image input (vision/multimodal) is supported; omitted = supported (not persisted). */
vision?: boolean;
/** Per-model max output tokens, a positive integer (wins over the Agent config); omitted = inherit the Agent value (the annotation is cleared). */
maxTokens?: number;
pricing?: ModelPricingDto;
/** Providing it overwrites and updates createdAt; omitting it keeps the existing value. */
apiKey?: string;