From 1e23452f484056d9d9ce383a9bb361b88faba719 Mon Sep 17 00:00:00 2001 From: Yaowei Zheng Date: Sun, 26 Jul 2026 16:31:51 +0800 Subject: [PATCH] Handoff-style /model switch command and per-turn thinking level (#64) Co-authored-by: Claude Fable 5 --- packages/cli/test/render.test.ts | 1 - packages/core/src/agent.ts | 27 +- packages/core/src/engine/context-engine.ts | 21 +- packages/core/src/interfaces.ts | 7 + packages/core/src/internal/session-title.ts | 8 +- packages/core/src/llm/generative-model.ts | 43 +- packages/core/src/omnimessage/types.ts | 8 +- packages/core/src/session.ts | 2 +- packages/core/test/agent.test.ts | 76 ++-- packages/core/test/compaction.test.ts | 1 - packages/core/test/engine.test.ts | 61 ++- packages/core/test/llm.test.ts | 76 +++- packages/core/test/replay.test.ts | 1 - packages/core/test/resume.test.ts | 44 +- packages/core/test/session-title.test.ts | 7 +- packages/core/test/trace.test.ts | 2 - packages/docs/content/agent-loop.en.md | 1 + packages/docs/content/agent-loop.zh.md | 1 + packages/docs/content/configuration.en.md | 2 +- packages/docs/content/configuration.zh.md | 2 +- packages/docs/content/interfaces.en.md | 4 +- packages/docs/content/interfaces.zh.md | 4 +- packages/docs/content/models.en.md | 4 +- packages/docs/content/models.zh.md | 4 +- packages/docs/content/omni-message.en.md | 3 +- packages/docs/content/omni-message.zh.md | 3 +- packages/docs/content/server-api.en.md | 9 +- packages/docs/content/server-api.zh.md | 8 +- .../docs/content/sessions-and-traces.en.md | 6 +- .../docs/content/sessions-and-traces.zh.md | 6 +- packages/docs/content/web-app.en.md | 2 +- packages/docs/content/web-app.zh.md | 2 +- packages/server/src/api/types.ts | 14 + packages/server/src/http/routes/sessions.ts | 22 +- .../server/src/runtime/session-manager.ts | 17 +- .../server/src/services/session-service.ts | 16 + .../server/test/agent-trace-detail.test.ts | 1 - packages/server/test/errors.test.ts | 1 - packages/server/test/session-index.test.ts | 55 ++- packages/server/test/session-loader.test.ts | 1 - packages/server/test/session-manager.test.ts | 25 +- packages/server/test/trace-service.test.ts | 1 - .../server/test/trace-subagent-expand.test.ts | 1 - packages/server/test/usage.test.ts | 1 - packages/web/e2e/draft.spec.mjs | 10 +- .../web/src/features/chat/agent-mentions.ts | 70 +++ packages/web/src/features/chat/chat-input.tsx | 401 ++++++++++++------ packages/web/src/features/chat/chat-page.tsx | 117 ++++- .../web/src/features/chat/handoff-banner.tsx | 45 +- .../web/src/features/chat/message-item.tsx | 11 +- .../web/src/features/chat/thinking-level.ts | 24 +- .../src/features/traces/trace-event-row.tsx | 7 +- packages/web/src/lib/omni/stream-model.ts | 18 +- packages/web/src/lib/strings-en.ts | 7 + packages/web/src/lib/strings.ts | 7 + packages/web/test/agent-mentions.test.ts | 41 +- packages/web/test/stream-model.test.ts | 29 -- packages/web/test/thinking-level.test.ts | 12 +- 58 files changed, 1064 insertions(+), 336 deletions(-) diff --git a/packages/cli/test/render.test.ts b/packages/cli/test/render.test.ts index de604e6..999f08e 100644 --- a/packages/cli/test/render.test.ts +++ b/packages/cli/test/render.test.ts @@ -194,7 +194,6 @@ describe("StreamRenderer", () => { model_context_window: 1, system_prompt: "sp", tools: [{ name: "exec_command", description: "test tool" }], - thinking_level: "medium", agent_state: "/a", workspace: "/w", }), diff --git a/packages/core/src/agent.ts b/packages/core/src/agent.ts index f58e25c..23a4e80 100644 --- a/packages/core/src/agent.ts +++ b/packages/core/src/agent.ts @@ -70,7 +70,7 @@ import type { ModelEntry } from "./state/index.js"; */ const MAX_SUBAGENT_DEPTH = 1; -/** The five valid thinking level names (session_meta additionally records the literal "default" for "no level"). */ +/** The five valid thinking level names (legacy session_meta additionally recorded the literal "default" for "no level"). */ const THINKING_LEVEL_NAMES: readonly ThinkingLevelName[] = [ "none", "low", @@ -80,9 +80,9 @@ const THINKING_LEVEL_NAMES: readonly ThinkingLevelName[] = [ ]; /** - * Narrows a recorded session_meta `thinking_level` back to a ThinkingLevelName; the literal - * "default" (no level recorded), a missing field (legacy Trace), and anything unknown are - * not levels and yield undefined. + * Narrows a legacy session_meta `thinking_level` back to a ThinkingLevelName; the literal + * "default" (no level recorded), a missing field (current Traces no longer record one), and + * anything unknown are not levels and yield undefined. */ function asThinkingLevelName(value: unknown): ThinkingLevelName | undefined { return typeof value === "string" && (THINKING_LEVEL_NAMES as readonly string[]).includes(value) @@ -291,6 +291,8 @@ export class Agent { }); return new Session({ + // session_meta holds per-session invariants only: the thinking level is a per-turn + // run parameter (RunOptions.thinkingLevel) and is deliberately not recorded here. meta: { session_id: sessionId, provider: modelEntry.provider, @@ -298,7 +300,6 @@ export class Agent { model_context_window: modelEntry.context_window ?? "unknown", system_prompt: systemPrompt, tools: rt.tools, - thinking_level: thinkingLevel ?? "default", agent_state: this.state.stateDir, workspace: workspaceDir, ...(opts.source !== undefined ? { source: opts.source } : {}), @@ -393,12 +394,15 @@ export class Agent { // original history); the vault uses current values (it's injected into the // subprocess environment, not the history, so a resumed Session should get the // latest keys too). - // The recorded thinking level is restored with the rest of session_meta (like model, - // Workspace, and system prompt): a resumed subagent session keeps its inherited level - // instead of re-reading this Agent's config. The literal "default" (no level recorded) - // — and a legacy Trace without the field — falls back to the Agent's current config. + // Back-compat: session_meta no longer records a thinking level (it became a per-turn + // run parameter), but OLD Traces still carry `thinking_level` in their meta JSON — when + // present, keep honoring it as this Session's default level (a resumed legacy subagent + // session keeps its inherited level instead of re-reading this Agent's config). The + // field is read loosely (it's gone from SessionMetaPayload); the legacy literal + // "default" — and any current Trace without the field — falls back to the Agent config. const thinkingLevel = - asThinkingLevelName(meta.thinking_level) ?? this.state.systemConfig.model?.thinking_level; + asThinkingLevelName((meta as unknown as Record).thinking_level) ?? + this.state.systemConfig.model?.thinking_level; const rt = await this.buildRuntime({ workspaceDir, @@ -438,6 +442,8 @@ export class Agent { }); return new Session({ + // Invariants only (the legacy thinking_level, when honored above, feeds the LLM default + // but is not re-recorded — new meta writes never contain it). meta: { session_id: sessionId, provider: modelEntry.provider, @@ -445,7 +451,6 @@ export class Agent { model_context_window: modelEntry.context_window ?? "unknown", system_prompt: meta.system_prompt, tools: rt.tools, - thinking_level: thinkingLevel ?? "default", agent_state: this.state.stateDir, workspace: workspaceDir, // The origin carries over from the original session_meta (a resumed scheduled/subagent diff --git a/packages/core/src/engine/context-engine.ts b/packages/core/src/engine/context-engine.ts index dbb4a77..3640791 100644 --- a/packages/core/src/engine/context-engine.ts +++ b/packages/core/src/engine/context-engine.ts @@ -58,7 +58,13 @@ import type { ToolCallOutputPayload, ToolCallPayload, } from "../omnimessage/index.js"; -import type { ApproveFn, EnvironmentInterface, LLMInterface, LLMOutcome } from "../interfaces.js"; +import type { + ApproveFn, + EnvironmentInterface, + LLMInterface, + LLMOutcome, + ThinkingLevelName, +} from "../interfaces.js"; /** Trace sink: `write` a complete/event/meta message; `rotate` starts a new file (compaction splits files). */ export interface TraceSink { @@ -96,6 +102,12 @@ export interface RunOptions { signal?: AbortSignal; /** Per-tool approval callback; defaults to denying everything (conservative, to avoid accidental approval when unattended). */ approve?: ApproveFn; + /** + * Thinking level for this run's LLM requests (a per-turn parameter): forwarded to every + * `streamGenerate` of this run — reconnect retries included; compaction requests keep the + * construction-time default (no override). Omitted = the LLM object's default. + */ + thinkingLevel?: ThinkingLevelName; } /** @@ -260,6 +272,9 @@ export class ContextEngine { const signal = opts?.signal; // Default approval policy: deny (conservative). CLI/Web will inject a real callback (interactive or permission-mode based). const approve: ApproveFn = opts?.approve ?? (async () => "deny"); + // Per-turn thinking level: applies to each of this run's LLM requests (reconnects included); + // compaction requests are out of scope and keep the LLM default. + const thinkingLevel = opts?.thinkingLevel; // Merge the Task-boundary compaction summary (the new context's first input, merged with // this Prompt), the carry-over left over from the last interruption, and this call's new @@ -329,7 +344,7 @@ export class ContextEngine { // Both LLM and Environment handle errors internally and guarantee a complete, closed // output with no thrown exceptions; the engine doesn't handle exceptions — // it decides retry/resend purely from `outcome`. - turn = yield* this.runTurn(attemptInput, approve, signal); + turn = yield* this.runTurn(attemptInput, approve, signal, thinkingLevel); // User interruption (the LLM stream was aborted, outcome=aborted, or `signal` fired // during tool execution): stop and hand control back to the user. @@ -517,6 +532,7 @@ export class ContextEngine { input: OmniMessage[], approve: ApproveFn, signal?: AbortSignal, + thinkingLevel?: ThinkingLevelName, ): AsyncGenerator { const queue = new MergeQueue(); // Tool outputs are collected in **completion order** (for streaming yield to the frontend); @@ -547,6 +563,7 @@ export class ContextEngine { const gen = this.llm.streamGenerate({ newMessages: input, ...(signal ? { signal } : {}), + ...(thinkingLevel !== undefined ? { thinkingLevel } : {}), }); for (;;) { const res = await gen.next(); diff --git a/packages/core/src/interfaces.ts b/packages/core/src/interfaces.ts index d36e3c2..0b40b58 100644 --- a/packages/core/src/interfaces.ts +++ b/packages/core/src/interfaces.ts @@ -103,6 +103,7 @@ export interface GenerativeModelConfig { contextWindow?: number; /** Output token cap per Request; non-positive (-1) means no explicit cap (omitted from the request). */ maxTokens?: number; + /** Construction-time default thinking level; a per-request `GenerativeModelParameters.thinkingLevel` overrides it for that request. */ thinkingLevel?: ThinkingLevelName; /** LLM Request timeout (ms): from system_config.model.timeoutMs; <=0 disables it. Defaults to 120000. */ requestTimeoutMs?: number; @@ -118,6 +119,12 @@ export interface GenerativeModelParameters { /** OmniMessage array for the input newly added this turn; implementations must merge it into a single UniMessage (multiple roles not accepted). */ newMessages: OmniMessage[]; signal?: AbortSignal; + /** + * Per-request thinking level override: applied to **this request only**; omitted falls back + * to the construction-time default (`GenerativeModelConfig.thinkingLevel`). The thinking + * level is a per-turn parameter, not a Session invariant. + */ + thinkingLevel?: ThinkingLevelName; } /** diff --git a/packages/core/src/internal/session-title.ts b/packages/core/src/internal/session-title.ts index bc6515b..e23ab67 100644 --- a/packages/core/src/internal/session-title.ts +++ b/packages/core/src/internal/session-title.ts @@ -29,11 +29,11 @@ const TITLE_MAX_CHARS = 30; /** * Special message markers that must never leak into a title. These are machine-inserted * XML-ish blocks (a skill invocation wraps the body in `…`, a - * subagent handoff / scheduled task prepend their own blocks) — meaningful to the runtime, - * noise in a title. The list is a fixed allowlist so ordinary angle-bracket text (e.g. a - * user pasting `
`) is left untouched. + * subagent handoff / scheduled task / model switch prepend their own blocks) — meaningful + * to the runtime, noise in a title. The list is a fixed allowlist so ordinary angle-bracket + * text (e.g. a user pasting `
`) is left untouched. */ -const MARKER_TAGS = ["use_skills", "handoff_from", "scheduled_task"]; +const MARKER_TAGS = ["use_skills", "handoff_from", "scheduled_task", "model_switch_from"]; /** * Strips machine-inserted marker blocks (see MARKER_TAGS) from conversation text so titles are diff --git a/packages/core/src/llm/generative-model.ts b/packages/core/src/llm/generative-model.ts index d261f72..bb88f46 100644 --- a/packages/core/src/llm/generative-model.ts +++ b/packages/core/src/llm/generative-model.ts @@ -863,6 +863,13 @@ export function isRetryableError(error: unknown): boolean { export class GenerativeModel implements LLMInterface { private readonly client: AutoLLMClient; private readonly uniConfig: UniConfig; + /** + * Construction-time default thinking level. Kept **out of the frozen uniConfig**: the + * effective level is resolved per request (`params.thinkingLevel ?? default`), so a turn can + * override it without rebuilding the model object — the thinking level is a per-turn + * parameter, not a Session invariant. + */ + private readonly defaultThinkingLevel: ThinkingLevelName | undefined; /** Streaming idle timeout (milliseconds); <= 0 disables it. A timeout is treated as needing reconnection. */ private readonly requestTimeoutMs: number; /** @@ -889,10 +896,23 @@ export class GenerativeModel implements LLMInterface { }); this.uniConfig = buildUniConfig(config); + this.defaultThinkingLevel = config.thinkingLevel; this.requestTimeoutMs = config.requestTimeoutMs ?? 120000; this.toolCallIds = config.toolCallIds ?? new ToolCallIdAllocator(); } + /** + * The UniConfig for one request: the shared frozen config plus this request's effective + * thinking level (per-request override, else the construction-time default; neither → the + * key stays off the wire, preserving the provider default). + */ + private requestConfig(override: ThinkingLevelName | undefined): UniConfig { + const thinking = mapThinkingLevel(override ?? this.defaultThinkingLevel); + return thinking === undefined + ? this.uniConfig + : { ...this.uniConfig, thinking_level: thinking }; + } + /** * Streaming generation (a single attempt, no internal retry). Merges * `params.newMessages` into one UniMessage to issue a stateful request, translating streamed @@ -967,7 +987,9 @@ export class GenerativeModel implements LLMInterface { // parse error) / aborted (user) / failed (other). null means it ended normally. let outcome: LLMOutcome | null = null; try { - const it = this.openStream(uniMessage, ac.signal)[Symbol.asyncIterator](); + const it = this.openStream(uniMessage, ac.signal, this.requestConfig(params.thinkingLevel))[ + Symbol.asyncIterator + ](); for (;;) { // The interruption check must happen **before pulling from upstream**: the user may // interrupt while this generator is suspended at the `yield` below (the typical case — @@ -1087,12 +1109,17 @@ export class GenerativeModel implements LLMInterface { * Opens the underlying AgentHub stream (a testing seam): defaults to * `streamingResponseStateful`; unit tests can subclass and override this method, feeding in a * controlled UniEvent stream to verify the outcome classification for timeout/network - * drop/interruption/error (without a real API). + * drop/interruption/error (without a real API). `config` is this request's resolved + * UniConfig (the shared frozen config plus the per-request thinking level). */ - protected openStream(uniMessage: UniMessage, signal: AbortSignal): AsyncIterable { + protected openStream( + uniMessage: UniMessage, + signal: AbortSignal, + config: UniConfig = this.uniConfig, + ): AsyncIterable { return this.client.streamingResponseStateful({ message: uniMessage, - config: this.uniConfig, + config, signal, }); } @@ -1133,6 +1160,10 @@ export function toolDefinitionsToSchemas(tools: ToolDefinition[]): ToolSchema[] * protocol equivalent. `tool_choice` is likewise never set — AgentHub only puts it on the wire * when UniConfig defines it, and leaving it off preserves the protocol default ("auto" when * tools are present). + * + * `thinkingLevel` is deliberately **not** baked in here: the effective level is resolved per + * request (`GenerativeModelParameters.thinkingLevel ?? the construction default`, see + * `GenerativeModel.requestConfig`), so a turn can override it on a live session. */ export function buildUniConfig(config: GenerativeModelConfig): UniConfig { const uniConfig: UniConfig = {}; @@ -1148,9 +1179,5 @@ export function buildUniConfig(config: GenerativeModelConfig): UniConfig { if (config.maxTokens !== undefined && config.maxTokens > 0) { uniConfig.max_tokens = config.maxTokens; } - const thinking = mapThinkingLevel(config.thinkingLevel); - if (thinking !== undefined) { - uniConfig.thinking_level = thinking; - } return uniConfig; } diff --git a/packages/core/src/omnimessage/types.ts b/packages/core/src/omnimessage/types.ts index bf0943e..2df80e7 100644 --- a/packages/core/src/omnimessage/types.ts +++ b/packages/core/src/omnimessage/types.ts @@ -75,6 +75,12 @@ export interface ToolDefinition { parameters?: Record; } +/** + * Session metadata. Holds **per-session invariants only** — values fixed for the Session's + * lifetime (model reference, assembled system prompt, tool schemas, paths, origin). Per-turn + * parameters (e.g. the thinking level, passed with each run) never belong here; legacy Traces + * may still carry a `thinking_level` field, which resume reads loosely for back-compat. + */ export interface SessionMetaPayload { session_id: string; /** The session model's provider group (paired with `model_id` to form a model reference). */ @@ -86,8 +92,6 @@ export interface SessionMetaPayload { system_prompt: string; /** The list of tool definitions this Session exposes to the model (full schema, matching what's sent to the LLM). */ tools: ToolDefinition[]; - /** The Session's effective thinking level (an explicit createSession option — e.g. subagent inheritance — else system_config.model.thinking_level; "default" when unconfigured). */ - thinking_level: string; /** Absolute path to the Agent State. */ agent_state: string; /** Absolute path to the Workspace. */ diff --git a/packages/core/src/session.ts b/packages/core/src/session.ts index 192ef24..ebad612 100644 --- a/packages/core/src/session.ts +++ b/packages/core/src/session.ts @@ -33,7 +33,7 @@ import type { } from "./engine/context-engine.js"; export interface SessionConfig { - /** Session metadata (session_id / provider / model_id / model_context_window / system_prompt / tools / thinking_level / agent_state / workspace). */ + /** Session metadata — per-session invariants only (session_id / provider / model_id / model_context_window / system_prompt / tools / agent_state / workspace / source). */ meta: SessionMetaPayload; llm: LLMInterface; environment: EnvironmentInterface; diff --git a/packages/core/test/agent.test.ts b/packages/core/test/agent.test.ts index 1310809..8456d2f 100644 --- a/packages/core/test/agent.test.ts +++ b/packages/core/test/agent.test.ts @@ -44,6 +44,21 @@ vi.mock("../src/environment/index.js", async (importOriginal) => { return { ...mod, Environment: CapturingEnvironment }; }); +// Captures every GenerativeModelConfig buildRuntime constructs: the effective thinking level +// is no longer observable through session_meta (it holds invariants only) — assertions read +// the construction default from the captured config instead. +const capturedLLMConfigs = vi.hoisted(() => ({ list: [] as { thinkingLevel?: string }[] })); +vi.mock("../src/llm/index.js", async (importOriginal) => { + const mod = await importOriginal(); + class CapturingGenerativeModel extends mod.GenerativeModel { + constructor(config: ConstructorParameters[0]) { + super(config); + capturedLLMConfigs.list.push(config as { thinkingLevel?: string }); + } + } + return { ...mod, GenerativeModel: CapturingGenerativeModel }; +}); + let tmpRoot: string; let prevHome: string | undefined; let restoreKeys: () => void; @@ -247,12 +262,13 @@ describe("Agent.createSession session source (session_meta origin marker)", () = }); describe("Agent.createSession thinking level (explicit option wins over the Agent config)", () => { - const uniThinkingOf = (llm: unknown): unknown => - ((llm as { uniConfig?: { thinking_level?: unknown } }).uniConfig ?? {}).thinking_level; - const llmOf = (session: unknown): unknown => - (session as { engine: { deps: { llm: unknown } } }).engine.deps.llm; + // The session's default level lives on the LLM object (per-request overrides fall back to + // it); session_meta holds per-session invariants only and never records a thinking level. + const defaultLevelOf = (session: unknown): unknown => + (session as { engine: { deps: { llm: { defaultThinkingLevel?: unknown } } } }).engine.deps.llm + .defaultThinkingLevel; - it("falls back to the Agent config for both the meta echo and the llm config", async () => { + it("falls back to the Agent config for the llm default; session_meta records no level", async () => { const agent = await createAgent(); // The seeded Agent config pins thinking_level "medium" — the only source when no option is given. expect(agent.state.systemConfig.model?.thinking_level).toBe("medium"); @@ -260,10 +276,12 @@ describe("Agent.createSession thinking level (explicit option wins over the Agen await fs.mkdir(ws, { recursive: true }); const session = await agent.createSession({ workspaceDir: ws }); try { - expect((session.metaMessage.payload as { thinking_level: string }).thinking_level).toBe( - "medium", - ); - expect(uniThinkingOf(llmOf(session))).toBe(mapThinkingLevel("medium")); + // session_meta contains ONLY per-session invariants: the thinking level is per-turn. + expect( + "thinking_level" in (session.metaMessage.payload as unknown as Record), + ).toBe(false); + expect(defaultLevelOf(session)).toBe("medium"); + expect(mapThinkingLevel("medium")).toBeDefined(); // the name maps onto the wire enum } finally { session.dispose(); } @@ -275,10 +293,7 @@ describe("Agent.createSession thinking level (explicit option wins over the Agen await fs.mkdir(ws, { recursive: true }); const session = await agent.createSession({ workspaceDir: ws, thinkingLevel: "high" }); try { - expect((session.metaMessage.payload as { thinking_level: string }).thinking_level).toBe( - "high", - ); - expect(uniThinkingOf(llmOf(session))).toBe(mapThinkingLevel("high")); + expect(defaultLevelOf(session)).toBe("high"); } finally { session.dispose(); } @@ -297,7 +312,9 @@ describe("run_subagent spawning follows the PARENT session (never the Project de /** * Spawns through the real runner and reads the child session_meta — the first message * handle.run yields, emitted before any LLM request; the run generator is closed right - * after, so nothing is ever sent upstream. + * after, so nothing is ever sent upstream. The child's effective thinking level is no + * longer in session_meta (invariants only): it's read from the captured LLM config the + * spawn constructed (the last one pushed). */ async function spawnedChildMeta( runner: SubagentRunner, @@ -305,11 +322,12 @@ describe("run_subagent spawning follows the PARENT session (never the Project de ): Promise<{ provider: string; model_id: string; - thinking_level: string; workspace: string; source?: string; + llm: { thinkingLevel?: string }; }> { const handle = await runner.spawn(input); + const llm = capturedLLMConfigs.list.at(-1)!; try { const gen = handle.run({ prompt: "noop" }); const first = await gen.next(); @@ -319,12 +337,14 @@ describe("run_subagent spawning follows the PARENT session (never the Project de // Child messages are stamped with the child Session id as the origin hop. expect(msg.origin?.[0]).toBe(handle.sessionId); await gen.return(undefined); - return msg.payload as { - provider: string; - model_id: string; - thinking_level: string; - workspace: string; - source?: string; + return { + ...(msg.payload as { + provider: string; + model_id: string; + workspace: string; + source?: string; + }), + llm, }; } finally { handle.dispose(); @@ -350,7 +370,7 @@ describe("run_subagent spawning follows the PARENT session (never the Project de const child = await spawnedChildMeta(runner, {}); expect(child.provider).toBe("anthropic"); expect(child.model_id).toBe("claude-sonnet-4-6"); - expect(child.thinking_level).toBe("high"); + expect(child.llm.thinkingLevel).toBe("high"); // Workspace inheritance (behavior that predates model/thinking inheritance): locked here. expect(child.workspace).toBe(ws); // The spawn site marks the child's own session_meta as subagent-created — the single @@ -379,7 +399,7 @@ describe("run_subagent spawning follows the PARENT session (never the Project de expect(child.provider).toBe("deepseek"); expect(child.model_id).toBe("deepseek-v4-pro"); // Thinking level and workspace are inherited implicitly even with an explicit model. - expect(child.thinking_level).toBe("medium"); + expect(child.llm.thinkingLevel).toBe("medium"); expect(child.workspace).toBe(ws); } finally { parent.dispose(); @@ -396,13 +416,13 @@ describe("run_subagent spawning follows the PARENT session (never the Project de const ws = path.join(tmpRoot, "ws-inherit-none"); await fs.mkdir(ws, { recursive: true }); const parent = await agent.createSession({ workspaceDir: ws }); + const parentLLM = capturedLLMConfigs.list.at(-1)!; const runner = lastSpawnedRunner(); try { - expect((parent.metaMessage.payload as { thinking_level: string }).thinking_level).toBe( - "default", - ); + expect("thinkingLevel" in parentLLM).toBe(false); const child = await spawnedChildMeta(runner, { agentId: "helper_agent" }); - expect(child.thinking_level).toBe("default"); + // The tri-state null reached the child: no level at all, not helper_agent's "medium". + expect("thinkingLevel" in child.llm).toBe(false); } finally { parent.dispose(); } @@ -447,7 +467,7 @@ describe("run_subagent spawning follows the PARENT session (never the Project de const child = await spawnedChildMeta(runner, { agentId: "helper_agent" }); expect(child.provider).toBe("anthropic"); expect(child.model_id).toBe("claude-sonnet-4-6"); - expect(child.thinking_level).toBe("xhigh"); + expect(child.llm.thinkingLevel).toBe("xhigh"); expect(child.workspace).toBe(ws); } finally { parent.dispose(); diff --git a/packages/core/test/compaction.test.ts b/packages/core/test/compaction.test.ts index 05d9514..0d5d86f 100644 --- a/packages/core/test/compaction.test.ts +++ b/packages/core/test/compaction.test.ts @@ -115,7 +115,6 @@ const metaMessage = sessionMeta({ model_context_window: 200000, system_prompt: "sp", tools: [], - thinking_level: "default", agent_state: "/tmp/state", workspace: "/tmp/ws", }); diff --git a/packages/core/test/engine.test.ts b/packages/core/test/engine.test.ts index dd7e712..cd902da 100644 --- a/packages/core/test/engine.test.ts +++ b/packages/core/test/engine.test.ts @@ -580,7 +580,6 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => { model_context_window: 1000, system_prompt: "sys", tools: [], - thinking_level: "medium", agent_state: "/root/p/worker/agent_state", workspace: "/tmp/w", }); @@ -666,6 +665,66 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => { ).toBe(false); expect(recorded.filter((m) => (m.payload as { text?: string }).text === "go")).toHaveLength(1); }); + + it("forwards RunOptions.thinkingLevel to every LLM request of the run (reconnects included); compaction keeps the default", async () => { + const levels: (string | undefined)[] = []; + let calls = 0; + const llm: LLMInterface = { + async *streamGenerate(params) { + calls += 1; + levels.push(params.thinkingLevel); + if (calls === 1) { + // First attempt drops: the reconnect retry must carry the same per-turn level. + yield assistantText("half", "timeout"); + return { status: "timeout" }; + } + if (calls === 2) { + // Retry completes with usage above the compaction threshold → a Task-boundary + // summarize compaction issues one more request (the engine's, not this run's turn): + // it must NOT carry the per-turn override. + yield assistantText("recovered"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 100, + }); + return { status: "completed" }; + } + yield assistantText("s"); + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 5, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 1, + reconnectBackoffMs: 1, + createLLM: () => llm, + compaction: { maxContextLength: 10, maxSessionTurns: -1, mode: "summarize", prompt: "SUM" }, + }); + + const all: OmniMessage[] = []; + for await (const msg of engine.run([userText("go")], { + approve: allowAll, + thinkingLevel: "high", + })) { + all.push(msg); + } + expect(calls).toBe(3); + // Turn attempt + reconnect retry carry the run's level; the compaction request does not. + expect(levels).toEqual(["high", "high", undefined]); + }); }); describe("ContextEngine async/incremental tool calls (overlapping execution)", () => { diff --git a/packages/core/test/llm.test.ts b/packages/core/test/llm.test.ts index a639221..0502146 100644 --- a/packages/core/test/llm.test.ts +++ b/packages/core/test/llm.test.ts @@ -15,8 +15,8 @@ import { ThinkingLevel, ToolCallArgumentParseError, } from "@prismshadow/agenthub"; -import type { UniEvent, UniMessage, UsageMetadata } from "@prismshadow/agenthub"; -import type { LLMOutcome } from "../src/interfaces.js"; +import type { UniConfig, UniEvent, UniMessage, UsageMetadata } from "@prismshadow/agenthub"; +import type { LLMOutcome, ThinkingLevelName } from "../src/interfaces.js"; import { EventTranslator, @@ -1034,7 +1034,7 @@ describe("config helpers", () => { expect("parameters" in schemas[1]!).toBe(false); }); - it("builds UniConfig with only provided fields", () => { + it("builds UniConfig with only provided fields (thinking level stays out — it is per-request)", () => { const cfg = buildUniConfig({ modelId: "claude-sonnet-4-6", tools: [{ name: "t", description: "d" }], @@ -1044,7 +1044,9 @@ describe("config helpers", () => { }); expect(cfg.system_prompt).toBe("You are concise."); expect(cfg.max_tokens).toBe(256); - expect(cfg.thinking_level).toBe(ThinkingLevel.HIGH); + // The thinking level is applied per request (override ?? construction default), never + // baked into the frozen config — see the request-config test below. + expect("thinking_level" in cfg).toBe(false); expect(cfg.tools).toEqual([{ name: "t", description: "d" }]); const minimal = buildUniConfig({ modelId: "m", tools: [] }); @@ -1181,6 +1183,72 @@ describe("isIncompleteStreamError", () => { }); }); +describe("GenerativeModel per-request thinking level", () => { + // Captures the UniConfig each request goes out with (the openStream seam now receives the + // per-request resolved config): the effective level = params.thinkingLevel ?? the + // construction default, mapped onto the wire enum; neither → the key stays off the wire. + function capturingModel(defaultLevel?: ThinkingLevelName): { + model: GenerativeModel; + configs: (UniConfig | undefined)[]; + } { + const configs: (UniConfig | undefined)[] = []; + class CapturingModel extends GenerativeModel { + protected override openStream( + _uni: UniMessage, + _signal: AbortSignal, + config?: UniConfig, + ): AsyncIterable { + configs.push(config); + return (async function* () { + yield ev({ + content_items: [{ type: "text", text: "ok" }], + finish_reason: "stop", + usage_metadata: { + cached_tokens: 0, + prompt_tokens: 1, + thoughts_tokens: 0, + response_tokens: 1, + }, + }); + })(); + } + } + const model = new CapturingModel({ + modelId: "claude-sonnet-4-6", + tools: [], + ...(defaultLevel !== undefined ? { thinkingLevel: defaultLevel } : {}), + }); + return { model, configs }; + } + + async function drainAll(gen: AsyncGenerator): Promise { + let res = await gen.next(); + while (!res.done) res = await gen.next(); + } + + it("applies the construction default when no override is given, per request", async () => { + const { model, configs } = capturingModel("medium"); + await drainAll(model.streamGenerate({ newMessages: [userText("hi")] })); + expect(configs[0]?.thinking_level).toBe(ThinkingLevel.MEDIUM); + }); + + it("a per-request override wins for that request only; the default returns afterwards", async () => { + const { model, configs } = capturingModel("medium"); + await drainAll(model.streamGenerate({ newMessages: [userText("a")], thinkingLevel: "high" })); + await drainAll(model.streamGenerate({ newMessages: [userText("b")] })); + expect(configs[0]?.thinking_level).toBe(ThinkingLevel.HIGH); + expect(configs[1]?.thinking_level).toBe(ThinkingLevel.MEDIUM); + }); + + it("no default and no override: thinking_level stays off the wire; an override still applies", async () => { + const { model, configs } = capturingModel(); + await drainAll(model.streamGenerate({ newMessages: [userText("a")] })); + await drainAll(model.streamGenerate({ newMessages: [userText("b")], thinkingLevel: "xhigh" })); + expect(configs[0] !== undefined && "thinking_level" in configs[0]).toBe(false); + expect(configs[1]?.thinking_level).toBe(ThinkingLevel.XHIGH); + }); +}); + describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () => { // Injects a controlled UniEvent stream through the protected openStream seam to verify the // outcome classification of timeout/network-drop/interrupt/error, without needing a real API. diff --git a/packages/core/test/replay.test.ts b/packages/core/test/replay.test.ts index 67850c9..544277d 100644 --- a/packages/core/test/replay.test.ts +++ b/packages/core/test/replay.test.ts @@ -50,7 +50,6 @@ function meta(): OmniMessage { model_context_window: 1000000, system_prompt: "SP", tools: [], - thinking_level: "default", agent_state: "/agent/state", workspace: "/ws", }); diff --git a/packages/core/test/resume.test.ts b/packages/core/test/resume.test.ts index 72a8eaa..69dbf1b 100644 --- a/packages/core/test/resume.test.ts +++ b/packages/core/test/resume.test.ts @@ -24,7 +24,7 @@ import { userText, } from "../src/omnimessage/index.js"; import type { OmniMessage, TokenCounts } from "../src/omnimessage/index.js"; -import { GenerativeModel, groupHistoryToUniMessages, mapThinkingLevel } from "../src/llm/index.js"; +import { GenerativeModel, groupHistoryToUniMessages } from "../src/llm/index.js"; import { readTrace } from "../src/trace/index.js"; import { tracesDir } from "../src/state/paths.js"; import { stubProviderKeys } from "./provider-keys.js"; @@ -90,7 +90,6 @@ function metaFor( model_context_window: 1000000, system_prompt: "ORIGINAL SYSTEM PROMPT", tools: [], - thinking_level: "default", agent_state: "/agent/state", workspace: workspaceDir, ...(source !== undefined ? { source } : {}), @@ -176,35 +175,40 @@ describe("agent.resumeSession", () => { ).toEqual(["text", "text", "abort"]); }); - it("restores the recorded thinking level; the literal 'default' falls back to the Agent config", async () => { - // A resumed subagent session must keep the level it inherited from its parent (recorded - // in session_meta, like model/Workspace/system prompt) — never silently re-read this - // Agent's own config. The seeded Agent config here pins "medium". + it("honors a legacy trace's recorded thinking_level; new meta never re-records it", async () => { + // session_meta no longer carries a thinking level (it became a per-turn run parameter), + // but OLD traces still have it in their meta JSON: resume must keep honoring it as the + // session's default level (a legacy subagent session keeps its inherited level instead of + // re-reading this Agent's config). The seeded Agent config here pins "medium". const agent = await createAgent({}); expect(agent.state.systemConfig.model?.thinking_level).toBe("medium"); const levelOf = (session: unknown): unknown => - ( - (session as { engine: { deps: { llm: { uniConfig?: { thinking_level?: unknown } } } } }) - .engine.deps.llm.uniConfig ?? {} - ).thinking_level; + (session as { engine: { deps: { llm: { defaultThinkingLevel?: unknown } } } }).engine.deps.llm + .defaultThinkingLevel; const recorded = metaFor(SID, workspace); - (recorded.payload as { thinking_level: string }).thinking_level = "xhigh"; + // A legacy trace: inject the retired field loosely into the on-disk meta JSON. + (recorded.payload as unknown as Record).thinking_level = "xhigh"; await writeTraceFile(tmpRoot, SID, [recorded, userText("hello")]); const inherited = await agent.resumeSession({ sessionId: SID }); - expect((inherited.metaMessage.payload as { thinking_level: string }).thinking_level).toBe( - "xhigh", - ); - expect(levelOf(inherited)).toBe(mapThinkingLevel("xhigh")); + expect(levelOf(inherited)).toBe("xhigh"); + // The rebuilt meta holds invariants only: the legacy field is honored but never re-recorded. + expect( + "thinking_level" in (inherited.metaMessage.payload as unknown as Record), + ).toBe(false); - // "default" records "no level": resume falls back to the Agent's current config, as before. + // A current trace (no field) — and the legacy literal "default" — fall back to the Agent config. const SID2 = "session-2026-07-06-11-00-00-abcdef02"; await writeTraceFile(tmpRoot, SID2, [metaFor(SID2, workspace), userText("hi")]); const fallback = await agent.resumeSession({ sessionId: SID2 }); - expect((fallback.metaMessage.payload as { thinking_level: string }).thinking_level).toBe( - "medium", - ); - expect(levelOf(fallback)).toBe(mapThinkingLevel("medium")); + expect(levelOf(fallback)).toBe("medium"); + + const SID3 = "session-2026-07-06-12-00-00-abcdef03"; + const legacyDefault = metaFor(SID3, workspace); + (legacyDefault.payload as unknown as Record).thinking_level = "default"; + await writeTraceFile(tmpRoot, SID3, [legacyDefault, userText("hi")]); + const viaDefault = await agent.resumeSession({ sessionId: SID3 }); + expect(levelOf(viaDefault)).toBe("medium"); }); it("does not write pairing placeholders to the trace file (resume is side-effect free)", async () => { diff --git a/packages/core/test/session-title.test.ts b/packages/core/test/session-title.test.ts index 13ce130..2632523 100644 --- a/packages/core/test/session-title.test.ts +++ b/packages/core/test/session-title.test.ts @@ -55,7 +55,6 @@ const META: SessionMetaPayload = { model_context_window: 1000, system_prompt: "sp", tools: [], - thinking_level: "default", agent_state: "/tmp/state", workspace: "/tmp/w", }; @@ -137,6 +136,12 @@ describe("session-title", () => { expect(stripConversationMarkers("data_analyst继续分析")).toBe( "继续分析", ); + // The /model switch origin block (the new session's first message) must not leak into the title either. + expect( + stripConversationMarkers( + "\nsession: session-01\ntrace: /t/x_001.jsonl\n\n继续这个任务", + ), + ).toBe("继续这个任务"); expect(stripConversationMarkers("render a
element")).toBe("render a
element"); }); diff --git a/packages/core/test/trace.test.ts b/packages/core/test/trace.test.ts index b690b1e..377acf5 100644 --- a/packages/core/test/trace.test.ts +++ b/packages/core/test/trace.test.ts @@ -25,7 +25,6 @@ function meta() { model_context_window: 200000, system_prompt: "test system prompt", tools: [{ name: "exec_command", description: "test tool" }], - thinking_level: "medium", agent_state: "/tmp/agent_state", workspace: "/tmp/workspace", }); @@ -87,7 +86,6 @@ describe("Writer", () => { model_context_window: 200000, system_prompt: "child prompt", tools: [], - thinking_level: "medium", agent_state: "/tmp/child_agent/agent_state", workspace: "/tmp/workspace", }); diff --git a/packages/docs/content/agent-loop.en.md b/packages/docs/content/agent-loop.en.md index ab3aa91..598f990 100644 --- a/packages/docs/content/agent-loop.en.md +++ b/packages/docs/content/agent-loop.en.md @@ -59,6 +59,7 @@ for await (const output of session.run([userText("Clean up the CSV files under d interface RunOptions { signal?: AbortSignal; // interrupt (e.g. Ctrl-C) approve?: ApproveFn; // per-tool approval; denies everything when omitted (conservative default) + thinkingLevel?: ThinkingLevelName; // this run's thinking level (per-turn, carried through reconnect retries; compaction requests keep the default) } ``` diff --git a/packages/docs/content/agent-loop.zh.md b/packages/docs/content/agent-loop.zh.md index d6d5193..07fcc12 100644 --- a/packages/docs/content/agent-loop.zh.md +++ b/packages/docs/content/agent-loop.zh.md @@ -56,6 +56,7 @@ for await (const output of session.run([userText("整理 data/ 下的 CSV 文件 interface RunOptions { signal?: AbortSignal; // 中断信号(如 Ctrl-C) approve?: ApproveFn; // 逐工具审批;未注入时默认全部拒绝(保守策略) + thinkingLevel?: ThinkingLevelName; // 本次 run 的思考等级(逐轮参数,覆盖到重连重试;压缩请求用默认值) } ``` diff --git a/packages/docs/content/configuration.en.md b/packages/docs/content/configuration.en.md index b91c02c..bb290a7 100644 --- a/packages/docs/content/configuration.en.md +++ b/packages/docs/content/configuration.en.md @@ -95,7 +95,7 @@ Edit this file via the CLI (`penguin config model …`) or the Web Models page | `system_prompt` | built-in template | Required; the only template with placeholder substitution | | `max_turns` | `100` | Maximum LLM turns per Task (-1 removes the cap) | | `model.max_tokens` | `32000` | Output Token limit per Request (-1 = no cap, provider default) | -| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh` | +| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh`; the session default, overridable per-Task | | `model.timeoutMs` | `120000` | Per-Request timeout (milliseconds) | | `compaction.max_context_length` | `128000` | Context Token threshold that triggers compaction | | `compaction.max_session_turns` | `-1` | Cumulative Session turn threshold (`-1` = unlimited) | diff --git a/packages/docs/content/configuration.zh.md b/packages/docs/content/configuration.zh.md index 351b6c7..746c44b 100644 --- a/packages/docs/content/configuration.zh.md +++ b/packages/docs/content/configuration.zh.md @@ -95,7 +95,7 @@ output = 0.857143 | `system_prompt` | 内置模板 | 必填;唯一进行占位符替换的模板 | | `max_turns` | `100` | 单个 Task 的最大 LLM 轮数(-1 不限制) | | `model.max_tokens` | `32000` | 单次输出 Token 上限(-1 不设上限,用服务商默认) | -| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh` | +| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh`;作为会话默认档位,可被逐轮 Task 参数覆盖 | | `model.timeoutMs` | `120000` | 单次 Request 超时(毫秒) | | `compaction.max_context_length` | `128000` | 触发压缩的上下文 Token 阈值 | | `compaction.max_session_turns` | `-1` | Session 累计轮数阈值(`-1` 不限制) | diff --git a/packages/docs/content/interfaces.en.md b/packages/docs/content/interfaces.en.md index 6c74917..4788f36 100644 --- a/packages/docs/content/interfaces.en.md +++ b/packages/docs/content/interfaces.en.md @@ -40,6 +40,7 @@ interface LLMInterface { interface GenerativeModelParameters { newMessages: OmniMessage[]; // only this turn's new messages (the impl owns history; mixed roles rejected) signal?: AbortSignal; + thinkingLevel?: ThinkingLevelName; // per-request override; omitted = the construction default } ``` @@ -78,7 +79,7 @@ interface GenerativeModelConfig { systemPrompt?: string; // fully assembled system prompt, placeholders substituted contextWindow?: number; maxTokens?: number; - thinkingLevel?: ThinkingLevelName; // "none" | "low" | "medium" | "high" | "xhigh" + thinkingLevel?: ThinkingLevelName; // construction default (a per-request parameter can override); "none" | "low" | "medium" | "high" | "xhigh" requestTimeoutMs?: number; // per-Request timeout, default 120000; <=0 disables toolCallIds?: ToolCallIdAllocator; // Session-level tool_call_id registry (pass the same instance across compaction) } @@ -179,6 +180,7 @@ session.run( interface RunOptions { signal?: AbortSignal; // interrupt (e.g. Ctrl-C) approve?: ApproveFn; // per-tool approval; denies everything when omitted + thinkingLevel?: ThinkingLevelName; // this run's thinking level (per-turn; compaction requests unaffected) } ``` diff --git a/packages/docs/content/interfaces.zh.md b/packages/docs/content/interfaces.zh.md index 2042825..7214d4f 100644 --- a/packages/docs/content/interfaces.zh.md +++ b/packages/docs/content/interfaces.zh.md @@ -40,6 +40,7 @@ interface LLMInterface { interface GenerativeModelParameters { newMessages: OmniMessage[]; // 仅本轮新增消息(实现自行维护历史,多 role 不接受) signal?: AbortSignal; + thinkingLevel?: ThinkingLevelName; // 本次请求的思考等级覆盖;缺省用构造默认值 } ``` @@ -78,7 +79,7 @@ interface GenerativeModelConfig { systemPrompt?: string; // 占位符替换完成后的完整系统提示词 contextWindow?: number; maxTokens?: number; - thinkingLevel?: ThinkingLevelName; // "none" | "low" | "medium" | "high" | "xhigh" + thinkingLevel?: ThinkingLevelName; // 构造期默认档位(逐请求参数可覆盖);"none" | "low" | "medium" | "high" | "xhigh" requestTimeoutMs?: number; // 单次 Request 超时,默认 120000;<=0 关闭 toolCallIds?: ToolCallIdAllocator; // Session 级 tool_call_id 唯一性登记表(压缩重建时传同一实例) } @@ -179,6 +180,7 @@ session.run( interface RunOptions { signal?: AbortSignal; // 中断信号(如 Ctrl-C) approve?: ApproveFn; // 逐工具审批;未注入时默认全部拒绝 + thinkingLevel?: ThinkingLevelName; // 本次 run 的思考等级(逐轮参数;压缩请求不受影响) } ``` diff --git a/packages/docs/content/models.en.md b/packages/docs/content/models.en.md index 3f54cb2..f7fa7d3 100644 --- a/packages/docs/content/models.en.md +++ b/packages/docs/content/models.en.md @@ -79,11 +79,11 @@ Some models in the preset catalog: deepseek-v4-pro / deepseek-v4-flash, gemini-3 ## Thinking levels -Five levels: `none | low | medium | high | xhigh`, configured per Agent as `model.thinking_level` in `system_config.yaml`, default medium. The Web pickers offer `low` and above only (many models cannot disable thinking; `none` stays a valid stored value and still displays). The chat draft view offers a quick picker next to the model selector: a picked level is written back to the selected Agent's setting immediately (the switched-to level becomes that Agent's new default and applies from the next session; a running session keeps the level it was created with). See [Configuration](/configuration). +Five levels: `none | low | medium | high | xhigh`, configured per Agent as `model.thinking_level` in `system_config.yaml`, default medium. The Web pickers offer `low` and above only (many models cannot disable thinking; `none` stays a valid stored value and still displays). The chat draft view offers a quick picker next to the model selector: a picked level is written back to the selected Agent's setting immediately (the switched-to level becomes that Agent's new default and applies from the next session). Inside an active session the thinking level is a **per-turn parameter**: the composer's picker lists only the levels and starts out showing the Agent config's level — while the user hasn't picked one it auto-follows the config (sends omit the level, so config edits keep taking effect); once picked, the level sticks for that session and rides on every subsequent send (it applies to that session's subsequent Tasks only and never writes back to the Agent config). See [Configuration](/configuration). ## Models decoupled from Agents -An Agent never binds a model: the model is chosen when a Session is created and stays locked for that Session; the same Agent can run different Sessions on different models. The three `pricing` buckets feed the usage/cost center's per-Token accounting. +An Agent never binds a model: the model is chosen when a Session is created and stays locked for that Session; the same Agent can run different Sessions on different models. The in-session `/model` command changes models handoff-style: it opens a new session for the same Agent on the new model, keeping the current Workspace, whose first message carries a `` source block (the source session id and its Trace file path) — the history is not injected into the new context (some models require thinking payloads and `fidelity` on history replay, which cannot cross models); the model reads the Trace file itself when it needs it, and the source session stays untouched. The three `pricing` buckets feed the usage/cost center's per-Token accounting. Credential handling: diff --git a/packages/docs/content/models.zh.md b/packages/docs/content/models.zh.md index bca618f..de722f2 100644 --- a/packages/docs/content/models.zh.md +++ b/packages/docs/content/models.zh.md @@ -79,11 +79,11 @@ api_key = "sk-..." ## 思考等级 -思考等级共五档:`none | low | medium | high | xhigh`,按 Agent 在 `system_config.yaml` 的 `model.thinking_level` 配置,默认 medium。Web 拾取器只提供 `low` 及以上档位(多数模型不支持关闭思考;`none` 仍是合法的已存值,能正常显示)。对话草稿页在模型选择器旁提供快捷拾取器:选定档位立即写回所选 Agent 的该项配置(切换后的档位即成为该 Agent 的新默认,自下一个 Session 生效;进行中的 Session 沿用创建时的档位)。见 [配置参考](/configuration)。 +思考等级共五档:`none | low | medium | high | xhigh`,按 Agent 在 `system_config.yaml` 的 `model.thinking_level` 配置,默认 medium。Web 拾取器只提供 `low` 及以上档位(多数模型不支持关闭思考;`none` 仍是合法的已存值,能正常显示)。对话草稿页在模型选择器旁提供快捷拾取器:选定档位立即写回所选 Agent 的该项配置(切换后的档位即成为该 Agent 的新默认,自下一个 Session 生效)。进行中的会话里,思考等级是**逐轮参数**:输入区拾取器只列出各档位,初始即显示 Agent 配置的档位——用户未手动选择时自动跟随配置下发(请求不携带档位,配置的修改持续生效);选定某档后即固定为该会话的档位,随之后每次发送携带(仅作用于该会话的后续 Task,不写回 Agent 配置)。见 [配置参考](/configuration)。 ## 模型与 Agent 解耦 -Agent 从不绑定模型:模型在创建 Session 时选定,并在该 Session 内锁定不变;同一个 Agent 可以在不同 Session 用不同模型运行。`pricing` 三档价格供用量/成本中心按 Token 计费。 +Agent 从不绑定模型:模型在创建 Session 时选定,并在该 Session 内锁定不变;同一个 Agent 可以在不同 Session 用不同模型运行。会话内的 `/model` 命令按 handoff 方式换模型:在同一 Agent 下新建一个使用新模型、沿用当前 Workspace 的会话,首条消息携带 `` 源块(源会话 id 与其 Trace 文件路径)——历史不注入新上下文(部分模型回放历史时必须携带 thinking 与 `fidelity`,不能跨模型),模型需要时按路径自行读取;原会话保持不变。`pricing` 三档价格供用量/成本中心按 Token 计费。 凭证处理: diff --git a/packages/docs/content/omni-message.en.md b/packages/docs/content/omni-message.en.md index ffb0248..22129bc 100644 --- a/packages/docs/content/omni-message.en.md +++ b/packages/docs/content/omni-message.en.md @@ -38,7 +38,6 @@ interface SessionMetaPayload { model_context_window: number | string; system_prompt: string; // fully assembled, placeholders substituted tools: ToolDefinition[]; // the complete tool schema sent to the model - thinking_level: string; // "default" when unconfigured agent_state: string; // absolute path of the Agent State workspace: string; // absolute path of the Workspace source?: "subagent" | "schedule"; // session origin; absent = user-created @@ -51,7 +50,7 @@ interface ToolDefinition { } ``` -On resume, the engine takes this Trace line as the runtime config — the model, system prompt and Workspace are immutable for the Session's lifetime. See [Sessions & Traces](/sessions-and-traces). +session_meta holds **per-session invariants only** — the model, system prompt and Workspace are immutable for the Session's lifetime; on resume, the engine takes this Trace line as the runtime config. See [Sessions & Traces](/sessions-and-traces). The thinking level is a per-turn parameter (sent with each Task) and is not recorded here; legacy Traces may still carry a `thinking_level` field in their meta, which resume keeps honoring for back-compat. ## model_msg: complete payloads diff --git a/packages/docs/content/omni-message.zh.md b/packages/docs/content/omni-message.zh.md index 9e1e6ec..c9a9b1b 100644 --- a/packages/docs/content/omni-message.zh.md +++ b/packages/docs/content/omni-message.zh.md @@ -38,7 +38,6 @@ interface SessionMetaPayload { model_context_window: number | string; system_prompt: string; // 占位符替换完成后的完整系统提示词 tools: ToolDefinition[]; // 发给模型的完整工具 schema - thinking_level: string; // 未配置时为 "default" agent_state: string; // Agent State 绝对路径 workspace: string; // Workspace 绝对路径 source?: "subagent" | "schedule"; // Session 来源;缺省 = 用户创建 @@ -51,7 +50,7 @@ interface ToolDefinition { } ``` -恢复 Session 时,引擎直接以 Trace 中的这条消息为运行时配置——模型、系统提示词、Workspace 在 Session 生命周期内不可变,见 [Session 与 Trace](/sessions-and-traces)。 +session_meta 只承载**会话级不变量**——模型、系统提示词、Workspace 在 Session 生命周期内不可变;恢复 Session 时引擎直接以 Trace 中的这条消息为运行时配置,见 [Session 与 Trace](/sessions-and-traces)。思考等级是逐轮参数(随每次 Task 下发),不记录在此;旧版 Trace 的 meta 里可能仍带 `thinking_level` 字段,恢复时按兼容逻辑继续生效。 ## model_msg:完整消息 diff --git a/packages/docs/content/server-api.en.md b/packages/docs/content/server-api.en.md index 7a4f701..4f90a8a 100644 --- a/packages/docs/content/server-api.en.md +++ b/packages/docs/content/server-api.en.md @@ -139,12 +139,12 @@ The paths below omit the `/api/sessions/:sessionId` prefix. For the storage mode | Method | Path | Description | | --- | --- | --- | -| GET | / | Session info | +| GET | / | Session info (the single-session GET additionally carries `tracePath`, the absolute path of the latest Trace file; list rows omit it) | | PATCH | / | Update: `{approvalMode?, archived?, title?}` | | DELETE | / | Delete the Session (along with its Traces and scratch files) | | GET | /messages | Full OmniMessage history | | GET | /stream | SSE event stream (next section) | -| POST | /tasks | Start a Task: `{input: TaskInputPart[]}` → 202 | +| POST | /tasks | Start a Task: `{input: TaskInputPart[], thinkingLevel?}` → 202 | | POST | /approvals/:toolCallId | Approval decision: `{decision}` is `allow` or `deny` → 204 | | POST | /abort | Interrupt the current Task: 202 when triggered, 204 when idle | | POST | /compact | Trigger context compaction: 202; 409 `nothing_to_compact` when there is nothing to compact | @@ -194,6 +194,9 @@ Key request bodies (explicit keys): // POST /api/sessions/:sessionId/tasks — start a Task interface TaskCreateRequest { input: TaskInputPart[]; + // Thinking level for this Task (a per-turn parameter, one of the five names; 400 otherwise); + // omitted = falls back to the Agent config + thinkingLevel?: "none" | "low" | "medium" | "high" | "xhigh"; } type TaskInputPart = | { type: "text"; text: string } @@ -205,6 +208,8 @@ interface ApprovalDecisionRequest { } ``` +The Web's `/model` switch has no dedicated endpoint: like the @ handoff, it composes the ordinary APIs above — session creation opens a new Session for the same Agent (the chosen model, the source Workspace carried over), then POST /tasks sends a first message opening with a `` source block (the source session id, its `tracePath`, the Workspace, and the previous model pair); the model reads that Trace file itself when it needs the earlier history. + ## Streaming (SSE) Real-time delivery uses Server-Sent Events, not WebSocket, on two channels (the ordering semantics of what the channels carry are on [Message Flow & Ordering](/message-flow)): diff --git a/packages/docs/content/server-api.zh.md b/packages/docs/content/server-api.zh.md index 7ad2fd9..f1a4c25 100644 --- a/packages/docs/content/server-api.zh.md +++ b/packages/docs/content/server-api.zh.md @@ -139,12 +139,12 @@ Schedule 写操作仅限 Owner。新建 Session 模式的任务,`modelId` 与 | 方法 | 路径 | 说明 | | --- | --- | --- | -| GET | / | Session 信息 | +| GET | / | Session 信息(单会话 GET 额外携带 `tracePath`:最新 Trace 文件的绝对路径;列表行不含) | | PATCH | / | 更新:`{approvalMode?, archived?, title?}` | | DELETE | / | 删除 Session(连同 Trace 与暂存文件) | | GET | /messages | 完整 OmniMessage 历史 | | GET | /stream | SSE 事件流(见下节) | -| POST | /tasks | 发起 Task:`{input: TaskInputPart[]}` → 202 | +| POST | /tasks | 发起 Task:`{input: TaskInputPart[], thinkingLevel?}` → 202 | | POST | /approvals/:toolCallId | 审批决定:`{decision}` 取 `allow` 或 `deny` → 204 | | POST | /abort | 中断当前 Task:已触发返回 202,无任务返回 204 | | POST | /compact | 触发上下文压缩:202;无可压缩内容返回 409 `nothing_to_compact` | @@ -194,6 +194,8 @@ GET /preview//<相对路径> (不鉴权,令牌即凭证 // POST /api/sessions/:sessionId/tasks —— 发起一个 Task interface TaskCreateRequest { input: TaskInputPart[]; + // 本次 Task 的思考等级(逐轮参数,五档之一;非法值 400);缺省 = 回退到 Agent 配置的档位 + thinkingLevel?: "none" | "low" | "medium" | "high" | "xhigh"; } type TaskInputPart = | { type: "text"; text: string } @@ -205,6 +207,8 @@ interface ApprovalDecisionRequest { } ``` +Web 的 `/model` 模型切换没有专用接口:它按 @ handoff 的方式复用上面的普通接口——先用会话创建接口在同一 Agent 下新建 Session(选定新模型并沿用源 Workspace),再 POST /tasks 发送以 `` 源块开头的首条消息(源会话 id、其 `tracePath`、Workspace 与原模型二元组),模型需要早前历史时自行读取该 Trace 文件。 + ## 流式接口(SSE) 实时通道采用 Server-Sent Events 而非 WebSocket,共两条(通道内承载的消息顺序语义见[消息流转与时序](/message-flow)): diff --git a/packages/docs/content/sessions-and-traces.en.md b/packages/docs/content/sessions-and-traces.en.md index 9f1290f..ac82a33 100644 --- a/packages/docs/content/sessions-and-traces.en.md +++ b/packages/docs/content/sessions-and-traces.en.md @@ -55,7 +55,7 @@ See `packages/core/src/trace/writer.ts` for the implementation. The head of a Trace (illustrative; one OmniMessage envelope per line): ```jsonl -{"timestamp":"2026-07-18T03:10:22.531Z","type":"session_meta","payload":{"session_id":"session-2026-07-18-11-10-22-3f8a1c2d","provider":"deepseek","model_id":"deepseek-v4-pro","model_context_window":1000000,"system_prompt":"…","tools":[…],"thinking_level":"medium","agent_state":"/home/u/.penguin/data/default_project/agents/default_agent/agent_state","workspace":"/home/u/work"}} +{"timestamp":"2026-07-18T03:10:22.531Z","type":"session_meta","payload":{"session_id":"session-2026-07-18-11-10-22-3f8a1c2d","provider":"deepseek","model_id":"deepseek-v4-pro","model_context_window":1000000,"system_prompt":"…","tools":[…],"agent_state":"/home/u/.penguin/data/default_project/agents/default_agent/agent_state","workspace":"/home/u/work"}} {"timestamp":"…","type":"event_msg","payload":{"type":"request_begin"}} {"timestamp":"…","type":"model_msg","payload":{"type":"text","role":"user","text":"Create hello.txt"}} {"timestamp":"…","type":"model_msg","payload":{"type":"tool_call","role":"assistant","name":"exec_command","arguments":"{\"cmd\":\"printf hi > hello.txt\"}","tool_call_id":"call_0"}} @@ -79,6 +79,10 @@ Recovery requires that the Workspace and the model still exist. What recovery gu Special case: if the latest Trace file ends with a completed compaction, that context is closed as a whole — resume starts from an empty context; in summarize mode the `` is reconstructed and prepended to the first input after resume. +## Model switch (/model) + +The Web's `/model` command changes models the way the @ handoff does: it creates a new Session under the same Agent via the ordinary session-creation API (the chosen model, **the source session's Workspace** — so files stay reachable), and the first message opens with a `` source block — the source session id, the absolute path of its latest Trace file, the Workspace, and the previous model pair — followed by whatever the user typed. The history is **not injected** into the new context: some models require thinking payloads and `fidelity` byte-for-byte when history is replayed, which cannot cross models — instead the model reads the source Trace file itself (JSONL, one message envelope per line) when it needs the earlier context. The source session and its Trace are untouched. + ## Field fidelity Each content message's opaque provider `fidelity` payload (thinking signatures, phase labels, encrypted reasoning, …) is preserved verbatim in the Trace and sent back verbatim — some models require it byte-for-byte on history replay, and any rewriting would break compatibility. This is one reason the Trace stores raw OmniMessage envelopes rather than a post-processed format. diff --git a/packages/docs/content/sessions-and-traces.zh.md b/packages/docs/content/sessions-and-traces.zh.md index 955e489..c19538d 100644 --- a/packages/docs/content/sessions-and-traces.zh.md +++ b/packages/docs/content/sessions-and-traces.zh.md @@ -55,7 +55,7 @@ Trace 是 append-only 的 JSON Lines 文件,每行一个 OmniMessage 信封( 一条 Trace 的开头(示意,每行一个 OmniMessage 信封): ```jsonl -{"timestamp":"2026-07-18T03:10:22.531Z","type":"session_meta","payload":{"session_id":"session-2026-07-18-11-10-22-3f8a1c2d","provider":"deepseek","model_id":"deepseek-v4-pro","model_context_window":1000000,"system_prompt":"…","tools":[…],"thinking_level":"medium","agent_state":"/home/u/.penguin/data/default_project/agents/default_agent/agent_state","workspace":"/home/u/work"}} +{"timestamp":"2026-07-18T03:10:22.531Z","type":"session_meta","payload":{"session_id":"session-2026-07-18-11-10-22-3f8a1c2d","provider":"deepseek","model_id":"deepseek-v4-pro","model_context_window":1000000,"system_prompt":"…","tools":[…],"agent_state":"/home/u/.penguin/data/default_project/agents/default_agent/agent_state","workspace":"/home/u/work"}} {"timestamp":"…","type":"event_msg","payload":{"type":"request_begin"}} {"timestamp":"…","type":"model_msg","payload":{"type":"text","role":"user","text":"创建 hello.txt"}} {"timestamp":"…","type":"model_msg","payload":{"type":"tool_call","role":"assistant","name":"exec_command","arguments":"{\"cmd\":\"printf hi > hello.txt\"}","tool_call_id":"call_0"}} @@ -79,6 +79,10 @@ Trace 是恢复的唯一事实来源,没有独立的会话数据库需要与 特殊情形:若最新 Trace 文件以一次完成的压缩收尾,则该上下文已整体关闭——恢复从空上下文开始;summarize 模式下会重建 `` 摘要,前置到恢复后第一轮输入中。 +## 模型切换(/model) + +Web 的 `/model` 命令按 @ handoff 的方式换模型:用普通的会话创建接口在同一 Agent 下新建一个 Session(选定新模型,**沿用源会话的 Workspace**,文件因此保持可达),首条消息以 `` 源块开头——携带源会话 id、其最新 Trace 文件的绝对路径、Workspace 与原模型二元组,用户输入的剩余文字紧随其后。历史**不注入**新上下文:部分模型回放历史时要求 thinking 与 `fidelity` 逐字一致,跨模型注入不可行——模型需要早前上下文时按路径自行读取源 Trace 文件(JSONL,每行一个消息信封)。源会话与其 Trace 不受任何影响。 + ## 字段保真 内容消息携带的不透明 Provider 保真负载 `fidelity`(思考签名、phase 分段标记、加密推理等)在 Trace 中原样保存、原样回传——部分模型在历史回放时要求该负载逐字一致,任何转写都会破坏兼容性。这也是 Trace 直接存储 OmniMessage 信封而非二次加工格式的原因之一。 diff --git a/packages/docs/content/web-app.en.md b/packages/docs/content/web-app.en.md index d029ee0..ddb952c 100644 --- a/packages/docs/content/web-app.en.md +++ b/packages/docs/content/web-app.en.md @@ -33,7 +33,7 @@ The interface language (中文 / English / system) and theme (light / dark / sys ### Creating a Conversation -A new conversation starts as a draft: pick the Agent, the Workspace (via a server-side directory browser), the approval mode, the model, and the thinking level before sending the first message. The Session is created on first send, and from then on its model and Workspace are locked. Switching the thinking level or the model makes the switched-to value the new default: the level is written back to the selected Agent's `model.thinking_level` immediately, and the picked model carries over as the next conversation's default; a running session keeps the level and model it was created with (the input area shows them read-only). +A new conversation starts as a draft: pick the Agent, the Workspace (via a server-side directory browser), the approval mode, the model, and the thinking level before sending the first message. The Session is created on first send, and from then on its model and Workspace are locked. Switching the thinking level or the model in the draft makes the switched-to value the new default: the level is written back to the selected Agent's `model.thinking_level` immediately, and the picked model carries over as the next conversation's default. Inside an active session the thinking level is a per-turn parameter: the composer's picker starts out showing the Agent config's level and auto-follows it until touched (sends omit the level, so config edits keep taking effect); a pick sticks for the session and rides on every subsequent send (never written back to the Agent config); the model stays locked per session — the `/model` command switches models the way the @ handoff does: it opens a new session for the same Agent on the picked model, keeping the current Workspace, whose first message carries a `` source block (source session id, Trace file path, previous model) followed by whatever was left in the composer (an interface-language auto-line when empty). In the new session that block collapses into a "switched model" banner linking back to the source conversation, and the model reads the source Trace file itself when it needs the earlier history. There are four approval modes: `allow-all`, `deny-all`, `read-only` (only read-only tools pass), and `always-ask`. See [Tools and Approvals](/tools). diff --git a/packages/docs/content/web-app.zh.md b/packages/docs/content/web-app.zh.md index 190cd9e..169d22c 100644 --- a/packages/docs/content/web-app.zh.md +++ b/packages/docs/content/web-app.zh.md @@ -33,7 +33,7 @@ penguin web ### 新建会话 -新会话从草稿开始:先选择 Agent、Workspace(服务器端目录浏览器选取)、审批模式、模型与思考等级,再发送第一条消息。Session 在首次发送时才真正创建,此后该会话的模型与 Workspace 即被锁定。切换思考等级或模型时,切换后的值即成为新的默认:思考等级立即写回所选 Agent 的 `model.thinking_level`,所选模型则作为下一个新会话的默认延续;进行中的会话沿用创建时的档位与模型(输入区只读展示)。 +新会话从草稿开始:先选择 Agent、Workspace(服务器端目录浏览器选取)、审批模式、模型与思考等级,再发送第一条消息。Session 在首次发送时才真正创建,此后该会话的模型与 Workspace 即被锁定。草稿里切换思考等级或模型时,切换后的值即成为新的默认:思考等级立即写回所选 Agent 的 `model.thinking_level`,所选模型则作为下一个新会话的默认延续。进行中的会话里,思考等级是逐轮参数:输入区拾取器初始显示 Agent 配置的档位并自动跟随(未选择时发送不携带档位,配置修改持续生效),选定后即固定为该会话档位、随每次发送下发(不写回 Agent 配置);模型仍在会话内锁定,改用 `/model` 命令切换模型——与 @ handoff 同一方式:在同一 Agent 下新建一个使用所选模型、沿用当前 Workspace 的会话并跳转,其首条消息携带 `` 源块(源会话 id、Trace 文件路径、原模型),输入框剩余文字随之发出(为空时发一句界面语言的自动消息);新会话中该源块折叠为一条“已切换模型”横幅,可点击回到原会话,模型需要早前历史时按路径自行读取源 Trace 文件。 审批模式共四种:`allow-all`(全部放行)、`deny-all`(全部拒绝)、`read-only`(仅放行只读工具)、`always-ask`(每次询问),详见[工具与审批](/tools)。 diff --git a/packages/server/src/api/types.ts b/packages/server/src/api/types.ts index c0a37df..7a49244 100644 --- a/packages/server/src/api/types.ts +++ b/packages/server/src/api/types.ts @@ -473,6 +473,14 @@ export interface SessionInfo { hasTrace: boolean; /** Whether archived (hidden from the default list, grouped under "Archived"). */ archived: boolean; + /** + * Absolute path of the session's latest Trace file (the current context shard); absent + * when no Trace exists yet. Populated on the **single-session GET only** — list rows omit + * it (locating it costs a directory walk per Session). The web's `/model` switch puts it + * into the new session's `` block so the model can read the source + * history itself when it needs it. + */ + tracePath?: string; } /** @@ -562,6 +570,12 @@ export type TaskInputPart = export interface TaskCreateRequest { input: TaskInputPart[]; + /** + * Thinking level for this Task's LLM requests (a per-turn parameter; one of + * `none | low | medium | high | xhigh`, anything else is a 400). Omitted = falls back to + * the session's default (the Agent config's `model.thinking_level`). + */ + thinkingLevel?: ThinkingLevelName; } export interface TaskCreateResponse { diff --git a/packages/server/src/http/routes/sessions.ts b/packages/server/src/http/routes/sessions.ts index ecbbe94..112b6e1 100644 --- a/packages/server/src/http/routes/sessions.ts +++ b/packages/server/src/http/routes/sessions.ts @@ -11,7 +11,7 @@ import path from "node:path"; import { Hono } from "hono"; import type { Context } from "hono"; import { imageUrlMessage, scratchpadDir, userText } from "@prismshadow/penguin-core"; -import type { OmniMessage } from "@prismshadow/penguin-core"; +import type { OmniMessage, ThinkingLevelName } from "@prismshadow/penguin-core"; import type { ApprovalMode, FilesStatResponse, @@ -58,6 +58,9 @@ const APPROVAL_MODES: readonly ApprovalMode[] = [ "always-ask", ]; +/** The five valid per-turn thinking level names (TaskCreateRequest.thinkingLevel). */ +const THINKING_LEVELS: readonly ThinkingLevelName[] = ["none", "low", "medium", "high", "xhigh"]; + /** Accepted `category` query values of the list endpoint (SessionCategory, spelled out for validation). */ const SESSION_CATEGORIES: readonly SessionCategory[] = [ "active", @@ -201,8 +204,13 @@ export function sessionsRoutes(deps: AppDeps): Hono { app.get("/:sessionId", async (c) => { const row = resolveSession(c); const hasTrace = await deps.sessionService.hasTrace(row); + const info = await deps.sessionService.toInfo(row, hasTrace); + // Single-session GET only: the latest Trace file's absolute path (a directory walk per + // call — too costly for list rows). The web's /model switch hands it to the new session's + // block so the model can read the source history itself. + const tracePath = hasTrace ? await deps.sessionService.latestTracePath(row) : undefined; return c.json({ - session: await deps.sessionService.toInfo(row, hasTrace), + session: { ...info, ...(tracePath !== undefined ? { tracePath } : {}) }, } satisfies SessionResponse); }); @@ -357,9 +365,15 @@ export function sessionsRoutes(deps: AppDeps): Hono { app.post("/:sessionId/tasks", async (c) => { const row = resolveSession(c); - const input = parseTaskInput(await readJson(c)); + const body = await readJson(c); + const input = parseTaskInput(body); + // Per-turn thinking level (optional): validated against the five names; omitted follows + // the session's default. + const thinkingLevel = optionalEnum(body, "thinkingLevel", THINKING_LEVELS); // 202: the Task executes on the server, decoupled from the SSE connection; sessionId is the current actual id (the new id after self-heal). - const { sessionId } = await deps.manager.startTask(row.sessionId, input); + const { sessionId } = await deps.manager.startTask(row.sessionId, input, { + ...(thinkingLevel !== undefined ? { thinkingLevel } : {}), + }); return c.json({ sessionId } satisfies TaskCreateResponse, 202); }); diff --git a/packages/server/src/runtime/session-manager.ts b/packages/server/src/runtime/session-manager.ts index 5b6ba23..cf3efa8 100644 --- a/packages/server/src/runtime/session-manager.ts +++ b/packages/server/src/runtime/session-manager.ts @@ -40,6 +40,7 @@ import type { SessionMetaPayload, SessionTitleResult, TextPayload, + ThinkingLevelName, } from "@prismshadow/penguin-core"; import type { ServerEvent, SessionStatus } from "../api/types.js"; import { HttpError, isMissingCredential, modelCredentialMissing } from "../http/errors.js"; @@ -70,7 +71,7 @@ export interface RuntimeSession { readonly sessionId: string; run( newMessages: OmniMessage[], - opts: { approve: ApproveFn; signal: AbortSignal }, + opts: { approve: ApproveFn; signal: AbortSignal; thinkingLevel?: ThinkingLevelName }, ): AsyncGenerator; compact(opts: { signal: AbortSignal }): AsyncGenerator; /** Whether compaction is possible and why; when not ok, compact() yields no messages (see core ContextEngine.compactability). */ @@ -379,8 +380,14 @@ export class SessionManager { * Start a Task: get-or-load → 409 * mutual-exclusion check → publish the input messages first → drive run in the * background. Returns the current actual session_id (the new id after self-heal). + * `opts.thinkingLevel` (optional, validated by the route) rides into this run's + * `session.run` options — a per-turn parameter, applied to this Task only. */ - async startTask(sessionId: string, input: OmniMessage[]): Promise<{ sessionId: string }> { + async startTask( + sessionId: string, + input: OmniMessage[], + opts?: { thinkingLevel?: ThinkingLevelName }, + ): Promise<{ sessionId: string }> { return this.withLock(sessionId, async () => { this.assertOpen(); this.assertAgentNotDeleting(sessionId); @@ -409,7 +416,11 @@ export class SessionManager { ...(pending.origin !== undefined ? { origin: pending.origin } : {}), }), }); - const gen = entry.session.run(input, { approve, signal: ac.signal }); + const gen = entry.session.run(input, { + approve, + signal: ac.signal, + ...(opts?.thinkingLevel !== undefined ? { thinkingLevel: opts.thinkingLevel } : {}), + }); // Title material is collected by the core Session itself during run; here we only // keep this call's input user text, used both as the "material present → attempt // generation" criterion and as the fallback title source if the LLM call fails. diff --git a/packages/server/src/services/session-service.ts b/packages/server/src/services/session-service.ts index d236635..d585b26 100644 --- a/packages/server/src/services/session-service.ts +++ b/packages/server/src/services/session-service.ts @@ -16,6 +16,7 @@ import path from "node:path"; import { open, readdir } from "node:fs/promises"; import { createAgent, + findLatestTraceFile, isSessionMeta, parseTraceLines, readTraceTolerant, @@ -389,6 +390,21 @@ export class SessionService { return this.toInfo(row, false); } + /** + * Absolute path of a Session's **latest** Trace file (the current context shard); + * undefined when no Trace exists. Costs a directory walk, so only the single-session + * GET includes it in the DTO (see SessionInfo.tracePath) — the web's `/model` switch + * hands it to the new session's `` block so the model can read the + * source history itself when it needs it. + */ + async latestTracePath(row: SessionRow): Promise { + const located = await findLatestTraceFile( + tracesDir(this.deps.root, row.projectId, row.agentId), + row.sessionId, + ); + return located?.path; + } + /** * One walk over the Trace directory: session_id → its **earliest** shard (the shard * whose head carries the original session_meta). Discovery (which Sessions have diff --git a/packages/server/test/agent-trace-detail.test.ts b/packages/server/test/agent-trace-detail.test.ts index 9585fe5..64b04ce 100644 --- a/packages/server/test/agent-trace-detail.test.ts +++ b/packages/server/test/agent-trace-detail.test.ts @@ -28,7 +28,6 @@ function metaPayload(): SessionMetaPayload { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: "/tmp/w", }; diff --git a/packages/server/test/errors.test.ts b/packages/server/test/errors.test.ts index 5c5fa78..d037e2f 100644 --- a/packages/server/test/errors.test.ts +++ b/packages/server/test/errors.test.ts @@ -360,7 +360,6 @@ describe("stream-error-watcher (LLM / Environment errors)", () => { model_context_window: 100000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: agentState, workspace: "/tmp/w", }), diff --git a/packages/server/test/session-index.test.ts b/packages/server/test/session-index.test.ts index 7d242b2..10a8f9c 100644 --- a/packages/server/test/session-index.test.ts +++ b/packages/server/test/session-index.test.ts @@ -149,7 +149,6 @@ describe("session-index", () => { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: "/tmp/w-restart", source: "subagent", @@ -176,7 +175,6 @@ describe("session-index", () => { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: "/tmp/w-cli", source: "schedule", @@ -281,7 +279,6 @@ describe("session-index", () => { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: "/tmp/w-sub", source: "subagent", @@ -390,7 +387,6 @@ describe("session-index", () => { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: "/tmp/cli-workspace", }; @@ -426,7 +422,6 @@ describe("session-index", () => { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: session.workspace, }; @@ -484,7 +479,6 @@ describe("session-index", () => { model_context_window: 1, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/a", workspace: "/w", }), @@ -539,4 +533,53 @@ describe("session-index", () => { ); expect(sessionIdCreatedAt("not-a-session")).toBeNull(); }); + + it("single-session GET exposes tracePath (the LATEST shard); absent without a trace; list rows omit it", async () => { + await configureModels(); + const res = await api.post(base(), {}); + const { session } = (await res.json()) as SessionCreateResponse; + // No trace yet: no tracePath. + const before = (await ( + await api.get(`/api/sessions/${session.sessionId}`) + ).json()) as SessionResponse; + expect(before.session.tracePath).toBeUndefined(); + + const meta: SessionMetaPayload = { + session_id: session.sessionId, + model_id: session.modelId, + provider: session.provider, + model_context_window: 128000, + system_prompt: "", + tools: [], + agent_state: "/tmp/a", + workspace: session.workspace, + }; + await writeTraceFile(t.root, projectId, "default_agent", "2026-07-02", session.sessionId, 1, [ + sessionMeta(meta), + userText("a"), + ]); + await writeTraceFile(t.root, projectId, "default_agent", "2026-07-03", session.sessionId, 2, [ + sessionMeta(meta), + userText("b"), + ]); + const after = (await ( + await api.get(`/api/sessions/${session.sessionId}`) + ).json()) as SessionResponse; + // The /model switch block hands this to the model: it must point at the latest shard. + expect(after.session.tracePath?.endsWith(`${session.sessionId}_002.jsonl`)).toBe(true); + // List rows omit it (locating it would cost a directory walk per row). + const list = (await (await api.get(base())).json()) as SessionsResponse; + expect(list.sessions.find((s) => s.sessionId === session.sessionId)?.tracePath).toBeUndefined(); + }); + + it("rejects an invalid per-task thinkingLevel with 400 (five names only)", async () => { + await configureModels(); + const res = await api.post(base(), {}); + const { session } = (await res.json()) as SessionCreateResponse; + const bad = await api.post(`/api/sessions/${session.sessionId}/tasks`, { + input: [{ type: "text", text: "hi" }], + thinkingLevel: "ultra", + }); + expect(bad.status).toBe(400); + }); }); diff --git a/packages/server/test/session-loader.test.ts b/packages/server/test/session-loader.test.ts index 1eb14cd..b6e876c 100644 --- a/packages/server/test/session-loader.test.ts +++ b/packages/server/test/session-loader.test.ts @@ -27,7 +27,6 @@ function meta(overrides: Partial = {}): SessionMetaPayload { model_context_window: 1000, system_prompt: "sp", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: path.join("/tmp", "does-not-exist-xyz"), ...overrides, diff --git a/packages/server/test/session-manager.test.ts b/packages/server/test/session-manager.test.ts index bde8d88..206c1e9 100644 --- a/packages/server/test/session-manager.test.ts +++ b/packages/server/test/session-manager.test.ts @@ -155,6 +155,29 @@ describe("session-manager", () => { }); }); + it("startTask forwards thinkingLevel into session.run options (per-turn, this Task only)", async () => { + sessions.updateApprovalMode("session-1", "allow-all"); + const seen: (string | undefined)[] = []; + const fake: RuntimeSession = { + sessionId: "session-1", + toolPermission: () => "rw", + generateTitle: async () => ({ title: null, usage: null }), + compactability: () => "ok" as const, + async *run(_input: OmniMessage[], opts: { thinkingLevel?: string }) { + seen.push(opts.thinkingLevel); + yield assistantText("ok"); + }, + async *compact(): AsyncGenerator {}, + }; + const manager = makeManager(loaderOf(fake)); + await manager.startTask("session-1", [userText("a")], { thinkingLevel: "high" }); + await waitFor(() => manager.statusOf("session-1") === "idle" && seen.length === 1); + // Omitted on the next Task: the session falls back to its default (nothing forwarded). + await manager.startTask("session-1", [userText("b")]); + await waitFor(() => manager.statusOf("session-1") === "idle" && seen.length === 2); + expect(seen).toEqual(["high", undefined]); + }); + it("LLM / tool failures in the message stream are persisted via drive (source=llm / environment, with the current Session context)", async () => { sessions.updateApprovalMode("session-1", "allow-all"); const captured: ErrorRecordArgs[] = []; @@ -276,7 +299,6 @@ describe("session-manager", () => { model_context_window: 1000, system_prompt: "sys", tools: [], - thinking_level: "default", agent_state: "/root/p1/child_agent/agent_state", workspace: "/tmp/w-child", }), @@ -691,7 +713,6 @@ describe("session-manager", () => { model_context_window: 1000, system_prompt: "sys", tools: [], - thinking_level: "default", agent_state: "/root/p1/child_agent/agent_state", workspace: "/tmp/w-child", }), diff --git a/packages/server/test/trace-service.test.ts b/packages/server/test/trace-service.test.ts index 13a1568..288f939 100644 --- a/packages/server/test/trace-service.test.ts +++ b/packages/server/test/trace-service.test.ts @@ -55,7 +55,6 @@ function metaPayload(): SessionMetaPayload { model_context_window: 1000, system_prompt: "sp", tools: [], - thinking_level: "default", agent_state: "/tmp/a", workspace: "/tmp/w", }; diff --git a/packages/server/test/trace-subagent-expand.test.ts b/packages/server/test/trace-subagent-expand.test.ts index 8cf16ed..d61a56a 100644 --- a/packages/server/test/trace-subagent-expand.test.ts +++ b/packages/server/test/trace-subagent-expand.test.ts @@ -39,7 +39,6 @@ function meta(sessionId: string, agentId: string): OmniMessage { model_context_window: 1000, system_prompt: "", tools: [], - thinking_level: "medium", agent_state: `/root/${PROJECT}/${agentId}/agent_state`, workspace: "/tmp/w", }; diff --git a/packages/server/test/usage.test.ts b/packages/server/test/usage.test.ts index 279b772..d6bc05e 100644 --- a/packages/server/test/usage.test.ts +++ b/packages/server/test/usage.test.ts @@ -38,7 +38,6 @@ function meta(sessionId: string, modelId: string, provider = "custom"): SessionM model_context_window: 100000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/tmp/x", workspace: "/tmp/w", }; diff --git a/packages/web/e2e/draft.spec.mjs b/packages/web/e2e/draft.spec.mjs index 8629ef6..c41d1fd 100644 --- a/packages/web/e2e/draft.spec.mjs +++ b/packages/web/e2e/draft.spec.mjs @@ -146,14 +146,16 @@ test("draft: pick model/approval -> reload restores them -> send creates the ses expect(first.session.provider).toBe("custom"); expect(first.session.approvalMode).toBe("read-only"); - // The written-through thinking level reached the session: its trace's session_meta records - // the level llmConfig was assembled with (per-session fixed), and the input area shows the - // read-only tag next to the locked model. + // session_meta holds per-session invariants only: the thinking level became a per-turn + // Task parameter and is no longer recorded in the trace meta. The session composer shows + // the editable per-turn picker instead of the old read-only tag: while the user hasn't + // picked, it displays the Agent config's level (auto-follow — "high" was written through + // above) and sends omit the level; a pick sticks and rides on every subsequent send. const replay = await ( await page.request.get(`${BASE}/api/sessions/${firstSessionId}/messages`) ).json(); const meta = replay.messages.find((m) => m.type === "session_meta"); - expect(meta?.payload?.thinking_level).toBe("high"); + expect(meta?.payload?.thinking_level).toBeUndefined(); await expect(page.getByTitle("思考等级:高")).toBeVisible(); // On a successful send the cache clears — except the model selection, which carries over as diff --git a/packages/web/src/features/chat/agent-mentions.ts b/packages/web/src/features/chat/agent-mentions.ts index 401e17d..29447d5 100644 --- a/packages/web/src/features/chat/agent-mentions.ts +++ b/packages/web/src/features/chat/agent-mentions.ts @@ -143,6 +143,76 @@ export function parseHandoffMessage(text: string): HandoffOrigin | null { return origin.agentId ? origin : null; } +/** + * Origin info for a `/model` switch new conversation (modeled on HandoffOrigin): the source + * Session, the absolute path of its latest Trace file (the model reads it for the earlier + * history — nothing is injected into the new context), the shared Workspace, and the + * previous model's paired reference. + */ +export interface ModelSwitchOrigin { + sessionId: string; + sessionTitle?: string; + /** Absolute path of the source session's latest Trace file (JSONL of OmniMessage envelopes). */ + tracePath?: string; + workspace?: string; + /** The source session's model reference pair (never concatenated — two separate fields). */ + prevProvider?: string; + prevModelId?: string; +} + +/** + * First message of a `/model` switch new conversation (in English, mirroring + * `handoffMessage`): the `` block states that this conversation continues + * an earlier one on a different model and carries the source Session / Trace path / + * Workspace / previous model pair. The earlier history is deliberately NOT injected — some + * models require thinking payloads and provider `fidelity` byte-for-byte when history is + * replayed, which cannot cross models — so the model reads the Trace file itself when it + * needs the context. When rendering, `parseModelSwitchMessage` collapses this into a + * one-line switch notice (the raw text isn't shown; the model still sees it as usual). + */ +export function modelSwitchMessage(origin: ModelSwitchOrigin): string { + const title = origin.sessionTitle ? ` (${origin.sessionTitle})` : ""; + const lines = [`session: ${origin.sessionId}${title}`]; + if (origin.tracePath) lines.push(`trace: ${origin.tracePath}`); + if (origin.workspace) lines.push(`workspace: ${origin.workspace}`); + if (origin.prevModelId) lines.push(`previous_model: ${origin.prevModelId}`); + if (origin.prevProvider) lines.push(`previous_provider: ${origin.prevProvider}`); + return [ + "", + "The user switched models: this conversation continues an earlier conversation, now on a different model. Its origin is listed below and the user's message, if any, follows. The earlier history is NOT in your context — when you need it, read the trace file at the path below (JSONL, one message envelope per line: user/assistant text, tool calls and results).", + ...lines, + "", + ].join("\n"); +} + +/** + * Inverse parse of `modelSwitchMessage` (lets the message stream collapse the origin block + * into a switch notice): returns origin info when the whole message is strictly one + * `` block, otherwise null (a normal user message renders as-is). + * Field lines parse as `key: value`; the session line's `(title)` label is split off; + * non-field lines such as the explanation sentence are ignored. + */ +export function parseModelSwitchMessage(text: string): ModelSwitchOrigin | null { + const block = /^\n([\s\S]*)\n<\/model_switch_from>$/.exec(text.trim()); + if (!block) return null; + const origin: ModelSwitchOrigin = { sessionId: "" }; + for (const line of block[1]!.split("\n")) { + const kv = /^(session|trace|workspace|previous_model|previous_provider): (.+)$/.exec(line); + if (!kv) continue; + const key = kv[1]!; + const value = kv[2]!; + if (key === "session") { + const labeled = /^([\w-]+) \((.*)\)$/.exec(value); + origin.sessionId = labeled ? labeled[1]! : value; + if (labeled?.[2] !== undefined) origin.sessionTitle = labeled[2]; + } else if (key === "trace") origin.tracePath = value; + else if (key === "workspace") origin.workspace = value; + else if (key === "previous_model") origin.prevModelId = value; + else origin.prevProvider = value; + } + return origin.sessionId ? origin : null; +} + /** Origin info for a scheduled-task trigger (the server's scheduledMessage `` block). */ export interface ScheduledOrigin { /** Task name (filename minus .toml). */ diff --git a/packages/web/src/features/chat/chat-input.tsx b/packages/web/src/features/chat/chat-input.tsx index dd8898f..d95a1d4 100644 --- a/packages/web/src/features/chat/chat-input.tsx +++ b/packages/web/src/features/chat/chat-input.tsx @@ -226,45 +226,37 @@ const NO_KEY_ICON = "M21 2l-2 2m-7.61 7.61a5.5 5.5 0 1 1-7.778 7.778 5.5 5.5 0 0 1 7.777-7.777zm0 0L15.5 7.5m0 0l3 3L22 7l-3-3m-3.5 3.5L19 4M2 2l20 20"; /** - * Model selector (draft state only; docked to the left of the send button): both the button and - * candidate items show the provider logo. The menu opens **downward** — the draft card is - * vertically centered with room below; a top quick-search box (reusing the model page's rule: - * filters by id / display name / provider name), with the candidate list capped by an internal - * scroll (max-h-56) so it never overflows the browser's viewport height no matter how many models - * there are. On narrow screens only the logo remains (name hidden); list items mark the project - * default. - * By default only models with a configured API key are listed (stored masked key — the same - * standard as the model page's key status; `envKey` is merely the NAME of a fallback env var and - * doesn't count), with the selected and the default model always visible even without a key; a - * muted bottom row reveals the remaining key-less models (marked by a struck-through key icon, - * with the "no key" text in its title) without closing the menu or changing the selection. When - * no model has a key at all, everything is listed directly. + * Model candidate panel (search box + grouped list + "show all" expander) shared by the + * draft-state ModelSelect dropdown and the in-session `/model` switch picker. Search and + * expanded state are internal and reset by remount (both hosts only render the panel while + * open); the list is capped by an internal scroll (max-h-56) so it never overflows the + * viewport no matter how many models there are. + * Dropdown order mirrors the model library page (visibleChatModels): a top quick-search box + * (the model page's rule — filters by id / display name / provider name); by default only + * models with a configured API key are listed (stored masked key — the same standard as the + * model page's key status; `envKey` is merely the NAME of a fallback env var and doesn't + * count), with the selected and the default model always visible even without a key; a muted + * bottom row reveals the remaining key-less models (marked by a struck-through key icon, with + * the "no key" text in its title) without closing the menu or changing the selection — when + * no model has a key at all, everything is listed directly. Rows carry the provider logo, the + * light-yellow "Free" badge for zero-cost models (same as the model library card), the + * project-default marker, and the selected checkmark. */ -function ModelSelect({ +function ModelMenuList({ models, value, defaultModel, - onChange, - disabled, + onPick, }: { models: ModelInfo[]; /** Currently selected (provider, modelId) pair; null = not yet chosen. */ value: ModelRefDto | null; defaultModel?: ModelRefDto; - onChange: (ref: ModelRefDto) => void; - disabled: boolean; + onPick: (m: ModelInfo) => void; }) { - const [open, setOpen] = useState(false); const [query, setQuery] = useState(""); - // Expanded "show all" state: collapses back to key-configured models on each open. + // Expanded "show all" state: collapses back to key-configured models on each open (remount). const [showAll, setShowAll] = useState(false); - const current = models.find((m) => sameModelRef(m, value)); - // Display rule matches the model page's card: display name, or falls back to the upstream id (grouping is already conveyed by the provider logo). - const label = current ? modelLabel(current) : (value?.modelId ?? "…"); - // Dropdown order mirrors the model library page: provider groups in MODEL_PROVIDERS order - // (user-defined groups after, custom last), in-group order preserved. By default the list - // keeps only key-configured models (selected/default always included; lists everything when - // no model has a key); the query filters what's visible. const visible = visibleChatModels(models, { showAll, query, selected: value, defaultModel }); // How many models the key filter hides under the current query (0 when expanded): drives the bottom "show all" row. const hiddenCount = showAll @@ -272,57 +264,7 @@ function ModelSelect({ : visibleChatModels(models, { showAll: true, query, selected: value, defaultModel }).length - visible.length; return ( - { - const next = !open; - setOpen(next); - if (next) { - // Each open starts from the unsearched, collapsed (configured-only) list. - setQuery(""); - setShowAll(false); - } - }} - className="flex h-8 max-w-44 shrink-0 items-center gap-1.5 rounded-md px-2 text-xs text-gray-500 transition-colors duration-150 hover:bg-gray-100 hover:text-gray-800 disabled:cursor-not-allowed disabled:opacity-50 dark:text-gray-400 dark:hover:bg-gray-800 dark:hover:text-gray-200" - > - - {/* When the card is narrower than @md, only the provider logo remains (title shows the full name). */} - {label} - - - - - } - > + <> {/* Quick search: supports model id / display name / provider name */}
{ - onChange({ provider: m.provider, modelId: m.modelId }); - setOpen(false); - }} + onClick={() => onPick(m)} className={`flex w-full items-center gap-2 px-3 py-1.5 text-left text-xs transition-colors duration-150 hover:bg-gray-100 dark:hover:bg-gray-800 ${ sameModelRef(m, value) ? "font-medium text-gray-900 dark:text-gray-100" @@ -398,6 +337,88 @@ function ModelSelect({
)} + + ); +} + +/** + * Model selector (draft state only; docked to the left of the send button): the button shows + * the provider logo + name (logo only when the card is narrower than @md; the title carries + * the full name), and the menu opens **downward** — the draft card is vertically centered + * with room below. The candidate list itself is the shared ModelMenuList panel (search, + * key-configured-first grouping, Free badge, "show all" expander — documented there). + */ +function ModelSelect({ + models, + value, + defaultModel, + onChange, + disabled, +}: { + models: ModelInfo[]; + /** Currently selected (provider, modelId) pair; null = not yet chosen. */ + value: ModelRefDto | null; + defaultModel?: ModelRefDto; + onChange: (ref: ModelRefDto) => void; + disabled: boolean; +}) { + const [open, setOpen] = useState(false); + const current = models.find((m) => sameModelRef(m, value)); + // Display rule matches the model page's card: display name, or falls back to the upstream id (grouping is already conveyed by the provider logo). + const label = current ? modelLabel(current) : (value?.modelId ?? "…"); + return ( + setOpen(!open)} + className="flex h-8 max-w-44 shrink-0 items-center gap-1.5 rounded-md px-2 text-xs text-gray-500 transition-colors duration-150 hover:bg-gray-100 hover:text-gray-800 disabled:cursor-not-allowed disabled:opacity-50 dark:text-gray-400 dark:hover:bg-gray-800 dark:hover:text-gray-200" + > + + {/* When the card is narrower than @md, only the provider logo remains (title shows the full name). */} + {label} + + + + + } + > + { + onChange({ provider: m.provider, modelId: m.modelId }); + setOpen(false); + }} + /> ); } @@ -406,25 +427,33 @@ function ModelSelect({ const SPARK_ICON = "M12 3l1.9 5.1L19 10l-5.1 1.9L12 17l-1.9-5.1L5 10l5.1-1.9L12 3z"; /** - * Conversation-time thinking-level picker (draft state only, docked left of the model - * selector): shows the **selected Agent's** current `model.thinking_level` and writes a picked - * level straight through to the Agent settings — llmConfig is assembled once per session, so - * the level applies to the session created on first send and becomes the Agent's new default - * (switch-becomes-default). Per review: a title bar names the control, and the menu lists - * the selectable levels with short names only (no descriptions, no "default" row, and no - * "none" — many models cannot disable thinking); an Agent without an explicit override shows - * an em dash until a level is picked, and a stored legacy "none" still displays via the - * label table (just never offered). + * Conversation-time thinking-level picker, used in two places. Both variants list only the + * concrete levels (per review: a title bar names the control; short names only, no + * descriptions, no "default"/"follow" row, and no "none" — many models cannot disable + * thinking; a stored legacy "none" still displays via the label table, just never offered): + * - Draft state (docked left of the model selector): shows the **selected Agent's** current + * `model.thinking_level` and writes a picked level straight through to the Agent settings — + * it applies to the session created on first send and becomes the Agent's new default + * (switch-becomes-default). An Agent without an explicit override shows an em dash until a + * level is picked. + * - Active session: the level is a **per-turn parameter** sent with each task. The displayed + * value initializes to the Agent config's level and auto-follows it while the user hasn't + * picked (the parent resolves the display value and keeps omitting the level from tasks + * until touched); an explicit pick sticks for the session and rides on every subsequent + * send, never writing through to the Agent config. */ function ThinkingLevelSelect({ value, onChange, disabled, + direction = "down", }: { - /** Current level ("" = no override yet); null = the Agent config is still loading. */ + /** Level to display and mark selected ("" = none to show yet); null = the Agent config is still loading (draft). */ value: string | null; onChange: (level: string) => void; disabled: boolean; + /** Popup direction: down for the draft card (room below), up for the bottom-docked session composer. */ + direction?: "down" | "up"; }) { const [open, setOpen] = useState(false); const label = @@ -433,7 +462,11 @@ function ThinkingLevelSelect({ } > - {/* Title bar: names the control (the rows themselves are just the five short names). */} + {/* Title bar: names the control (the rows themselves are just the short names). */}
{S.chat.thinkingLevel}
@@ -706,10 +739,12 @@ export function ChatInput({ modelRef, models, onChangeModel, + onSwitchModel, defaultModel, thinkingLevel, onChangeThinkingLevel, - sessionThinkingLevel, + turnThinkingLevel, + onChangeTurnThinkingLevel, contextWindow, contextNow, contextStale = false, @@ -749,6 +784,14 @@ export function ChatInput({ models?: ModelInfo[]; /** Changes the selected model in draft state; no longer passed once the Session is created and the model is locked. */ onChangeModel?: (ref: ModelRefDto) => void; + /** + * Session state: model switch via the `/model` command — forks the session onto the picked + * model (a NEW session carrying this conversation) and navigates there; any text remaining + * after the command token is posted as the new session's first task. Returns whether it + * succeeded (draft kept on failure). Only passed for an active session (the command is + * additionally gated on not running/compacting); picking the current model is a no-op. + */ + onSwitchModel?: (ref: ModelRefDto, input: TaskInputPart[]) => Promise; /** Project default model (marked "default" on the selector's candidate item). */ defaultModel?: ModelRefDto; /** @@ -764,11 +807,17 @@ export function ChatInput({ */ onChangeThinkingLevel?: (level: string) => void; /** - * Session state: the session's fixed thinking level (captured from session_meta on replay; - * llmConfig is assembled once per session, so it cannot change mid-session). Rendered as a - * read-only tag next to the locked model; null/undefined = unknown (nothing shown). + * Session state: the per-turn thinking level to DISPLAY — the parent resolves it as "the + * user's pick for this session, else the Agent config's level" ("" = neither known yet), + * so the picker auto-follows the config until touched. The send path is the parent's own + * state: while untouched nothing is sent with tasks (the server/core fallback applies and + * mid-session Agent-config edits keep taking effect); an explicit pick sticks and is sent + * with every subsequent task, never writing through to the Agent config (that behavior + * stays draft-only). */ - sessionThinkingLevel?: string | null; + turnThinkingLevel?: string; + /** Session state: pins the per-turn thinking level for this session; also enables the editable picker. */ + onChangeTurnThinkingLevel?: (level: string) => void; /** Model's context window (from models config; when not configured, the ring's cap falls back to 128000 via resolveContextWindow). */ contextWindow?: number; /** Current context usage (total of the most recent main-session Request). */ @@ -824,6 +873,13 @@ export function ChatInput({ const [slashIndex, setSlashIndex] = useState(0); // Slash token start where Escape closed the menu (mirrors mentionDismissed: the menu stays shut for that one token). const [slashDismissed, setSlashDismissed] = useState(null); + // /model switch picker (session state): opened by the /model command. The command consumes + // its slash token immediately (same as /compact), so closing the picker — Escape, click + // outside, or the picked-current-model no-op — can never re-open the slash menu, and there + // is no stale token range to recompute at pick time; whatever text remains is the draft + // (and becomes the new session's first message on a successful pick). + const [modelSwitchOpen, setModelSwitchOpen] = useState(false); + const modelSwitchRef = useRef(null); // Anchor for the popups that open upward, and the room actually available above them. const anchorRef = useRef(null); const [upwardMaxH, setUpwardMaxH] = useState(); @@ -882,6 +938,22 @@ export function ChatInput({ void onCompact(); }, }, + // Model switch (active idle session only — the parent passes onSwitchModel just there; + // draft state has its own model picker). Gated on the model list being loaded: without + // it the picker would open empty. Running the command consumes the /model token (like + // /compact) and opens the picker; the rest of the draft stays. + ...(onSwitchModel && models && models.length > 0 + ? [ + { + cmd: "/model", + desc: S.chat.switchModel, + run: () => { + clearInput(); + setModelSwitchOpen(true); + }, + }, + ] + : []), // Each installed skill gets its own entry: `/` toggles that skill's selection (without sending), description follows the UI language. ...skillSlashItems(skills, locale).map((s) => ({ cmd: s.cmd, @@ -892,11 +964,12 @@ export function ChatInput({ }, })), ]; - }, [onCompact, onTextChange, skills, locale, toggleSkill]); + }, [onCompact, onSwitchModel, models, onTextChange, skills, locale, toggleSkill]); // Positional matching (like @ mentions): a slash opens the menu from any caret position; // running a command removes just the token, leaving the rest of the text intact. Doesn't - // reopen after Escape until the caret sits on a different token. - const slashTok = !running && !compacting ? matchSlash(text, caret) : null; + // reopen after Escape until the caret sits on a different token; suppressed while the + // /model picker is open (its trigger token is still in the text). + const slashTok = !running && !compacting && !modelSwitchOpen ? matchSlash(text, caret) : null; slashMatchRef.current = slashTok; const slashMatches = slashTok && slashTok.start !== slashDismissed @@ -905,19 +978,81 @@ export function ChatInput({ const slashOpen = slashMatches.length > 0; const activeSlash = slashMatches[Math.min(slashIndex, slashMatches.length - 1)]; - // @ subagent menu: the `@prefix` currently being typed at the cursor drives candidate filtering (slash menu takes priority; doesn't reopen after Escape). - const mention = !running && !compacting && !slashOpen ? matchMention(text, caret) : null; + // @ subagent menu: the `@prefix` currently being typed at the cursor drives candidate + // filtering (the slash menu and the /model picker take priority; doesn't reopen after Escape). + const mention = + !running && !compacting && !slashOpen && !modelSwitchOpen ? matchMention(text, caret) : null; const mentionMatches = mention && mention.start !== mentionDismissed ? filterAgents(agents, mention.query) : []; const mentionOpen = mentionMatches.length > 0; const activeMention = mentionMatches[Math.min(mentionIndex, mentionMatches.length - 1)]; - // Both menus above are drawn upward (`bottom-full`) from the composer, so their ceiling is + // Close the /model picker on click-outside / Escape (same convention as Dropdown; the + // panel has no trigger button of its own, so the handling lives here). + useEffect(() => { + if (!modelSwitchOpen) return; + // globalThis.* event types: the React ones imported above would shadow the DOM ones here. + const onClick = (e: globalThis.MouseEvent) => { + if (modelSwitchRef.current && !modelSwitchRef.current.contains(e.target as Node)) { + setModelSwitchOpen(false); + } + }; + const onKey = (e: globalThis.KeyboardEvent) => { + if (e.key === "Escape") setModelSwitchOpen(false); + }; + window.addEventListener("mousedown", onClick); + window.addEventListener("keydown", onKey); + return () => { + window.removeEventListener("mousedown", onClick); + window.removeEventListener("keydown", onKey); + }; + }, [modelSwitchOpen]); + + /** + * /model pick: the CURRENT model is a no-op (close only), and the run state is re-checked + * — the picker may have survived a status flip (a task/compaction starting while it was + * open). Otherwise the pick opens a new session on the chosen model via onSwitchModel, + * with the first-task input assembled **like a normal send**: the remaining draft text + * (wrapped with the selected skills; an interface-language auto-line when empty — same + * convention as the skills auto message) plus the attached images. On failure the draft + * is kept so the user can retry. + */ + const pickSwitchModel = async (m: ModelInfo) => { + setModelSwitchOpen(false); + if (!onSwitchModel || busy || running || compacting) return; + if (sameModelRef(m, modelRef)) return; + const rest = textRef.current.trim(); + const bodyText = + rest || + (selectedSkills.length > 0 + ? S.chat.skillsAutoMessage(selectedSkills) + : S.chat.modelSwitchAutoMessage); + const body = buildSkillsMessage(selectedSkills, bodyText); + const input: TaskInputPart[] = [{ type: "text", text: body }]; + for (const url of images) input.push({ type: "image_url", imageUrl: url }); + setBusy(true); + try { + const ok = await onSwitchModel({ provider: m.provider, modelId: m.modelId }, input); + if (ok) { + // Consumed into the new session's first task (the parent has already discarded the + // draft cache — no change callbacks here, same as send()). + setText(""); + setImages([]); + setSelectedSkills([]); + requestAnimationFrame(autoGrow); + } + } finally { + setBusy(false); + textareaRef.current?.focus(); + } + }; + + // The menus above are drawn upward (`bottom-full`) from the composer, so their ceiling is // whatever ancestor clips overflow — on the draft page that's the centered scroll area, whose // top edge sits well below the viewport's. A static `40vh` cap can't know that distance and // clipped the first rows on shorter windows, so measure the real gap when a menu opens. useEffect(() => { - if (!slashOpen && !mentionOpen) return; + if (!slashOpen && !mentionOpen && !modelSwitchOpen) return; const measure = () => { const el = anchorRef.current; if (!el) return; @@ -935,7 +1070,7 @@ export function ChatInput({ measure(); window.addEventListener("resize", measure); return () => window.removeEventListener("resize", measure); - }, [slashOpen, mentionOpen]); + }, [slashOpen, mentionOpen, modelSwitchOpen]); /** Auto-grow the textarea (caps at roughly 6 lines, scrolls internally beyond that). */ const autoGrow = () => { @@ -1184,6 +1319,29 @@ export function ChatInput({
)} + {/* /model switch picker (session state): reuses the draft model dropdown's list — + search + configured-key-first grouping + "show all"; the current model is marked and + picking it is a no-op. The /model token was already consumed when the command ran, + so cancelling (Escape / click outside) keeps the remaining draft and cannot re-open + the slash menu. */} + {modelSwitchOpen && models && ( +
+
+ {S.chat.switchModelTitle} +
+ void pickSwitchModel(m)} + /> +
+ )} + {/* @ subagent menu (triggered by typing @; interaction matches the slash menu) */} {mentionOpen && (
)} - {/* Session state: the session's fixed thinking level (from session_meta), read-only next - to the locked model; hidden when it isn't one of the five levels (e.g. "default"). */} - {!onChangeModel && - (() => { - const label = thinkingLevelLabel(S.chat.thinkingLevelNames, sessionThinkingLevel); - return ( - label && ( - - - {label} - - ) - ); - })()} + {/* Session state: per-turn thinking level (editable) — displays the user's pick, + else the Agent config's level (auto-follow; the parent resolves it). While + untouched nothing rides on tasks; a pick sticks for the session and is sent + with every subsequent send, never writing through to the Agent config. */} + {!onChangeModel && onChangeTurnThinkingLevel && ( + + )} {/* Left of the send button: model selector in draft state; once the Session is created the model is locked, shown read-only (still with the provider logo). */} {models && onChangeModel ? ( (null); + // Per-turn thinking level, local per-session UI state: "" = untouched — the picker then + // displays the Agent config's level and postTask omits thinkingLevel (auto-follow: the + // server/core fallback applies, so mid-session Agent-config edits keep taking effect). + // Once the user picks a level it sticks for the session and rides on every subsequent + // postTask. Never written through to the Agent config (that behavior stays draft-only). + const [turnThinkingLevel, setTurnThinkingLevel] = useState(""); const routeSessionId = params.sessionId ?? null; const filesPanel = useFilesPanel(routeSessionId); @@ -183,6 +191,27 @@ export function ChatPage() { }; }, [projectId, selectedAgentId]); + // The session Agent's configured thinking level ("" = unset/loading), via the same + // agent-config endpoint the draft picker uses: the in-session picker DISPLAYS this while + // the user hasn't picked a level (auto-follow — sending still omits the level until + // touched, see turnThinkingLevel). Refetched when the session's Agent changes; a failed + // fetch leaves it unset (the picker then shows an em dash until picked). + const [agentThinkingLevel, setAgentThinkingLevel] = useState(""); + useEffect(() => { + setAgentThinkingLevel(""); + if (!projectId || !selectedAgentId) return; + let cancelled = false; + api + .getAgentConfig(projectId, selectedAgentId) + .then((res) => { + if (!cancelled) setAgentThinkingLevel(res.config.model?.thinkingLevel ?? ""); + }) + .catch(() => undefined); + return () => { + cancelled = true; + }; + }, [projectId, selectedAgentId]); + // The Session list is paged: a deep-linked Session (old bookmark, cross-page jump) may sit // beyond the loaded pages. Look it up directly and insert it before the auto-select effect // below concludes it doesn't exist; only a failed probe releases that redirect. @@ -240,11 +269,13 @@ export function ChatPage() { // concurrent mounts only issues one files/stat call. const statCacheRef = useRef(new Map>()); - // Session switch: resets the cost and the file-card existence cache, avoiding stale data from - // the previous Session (Files panel state resets itself keyed on sessionId inside use-files-panel). + // Session switch: resets the cost, the file-card existence cache, and the per-turn thinking + // level (it's per-session UI state), avoiding stale data from the previous Session (Files + // panel state resets itself keyed on sessionId inside use-files-panel). useEffect(() => { setSessionCost(null); setCostUncosted(false); + setTurnThinkingLevel(""); statCacheRef.current = new Map(); }, [routeSessionId]); @@ -353,7 +384,15 @@ export function ChatPage() { async (input: TaskInputPart[]): Promise => { if (!selected) return false; try { - const res = await api.postTask(selected.sessionId, { input }); + // An explicitly picked per-turn thinking level rides on each task; "" (untouched) + // sends nothing — the server/core falls back to the Agent config, so config edits + // keep taking effect mid-session until the user pins a level. + const res = await api.postTask(selected.sessionId, { + input, + ...(turnThinkingLevel + ? { thinkingLevel: turnThinkingLevel as TaskCreateRequest["thinkingLevel"] } + : {}), + }); discardSessionDraft(); await syncHealedSessionId(selected.sessionId, res.sessionId); return true; @@ -363,7 +402,62 @@ export function ChatPage() { return false; } }, - [selected, discardSessionDraft, syncHealedSessionId], + [selected, turnThinkingLevel, discardSessionDraft, syncHealedSessionId], + ); + + // /model switch (handoff-style, mirroring onHandoff exactly): opens a NEW session for the + // SAME agent on the picked model via the normal createSession API — deliberately with the + // SOURCE session's Workspace, so files the conversation refers to stay reachable — then + // posts a first task whose input starts with a source block (source + // session id / its latest trace path / workspace / previous model pair) followed by the + // user's remainder text and images. The earlier history is NOT injected into the new + // context (some models require thinking payloads and provider fidelity byte-for-byte on + // history replay, which cannot cross models); the model reads the source trace file itself + // when it needs the context. Returns false on failure, keeping the draft so it can be + // resent (the empty session that never got its first message is deleted, like handoff). + const onSwitchModel = useCallback( + async (ref: ModelRefDto, input: TaskInputPart[]): Promise => { + if (!projectId || !selected) return false; + // The source's latest trace file path comes from the single-session GET (list rows + // don't carry it); best-effort — a brand-new source has no trace, the block then + // simply omits the line. + const tracePath = await api + .getSession(selected.sessionId) + .then((res) => res.session.tracePath) + .catch(() => undefined); + const origin: TaskInputPart = { + type: "text", + text: modelSwitchMessage({ + sessionId: selected.sessionId, + ...(selected.title !== undefined ? { sessionTitle: selected.title } : {}), + ...(tracePath !== undefined ? { tracePath } : {}), + workspace: selected.workspace, + prevProvider: selected.provider, + prevModelId: selected.modelId, + }), + }; + let createdId: string | null = null; + try { + const created = await api.createSession(projectId, selected.agentId, { + provider: ref.provider, + modelId: ref.modelId, + workspace: selected.workspace, + approvalMode: selected.approvalMode, + }); + createdId = created.session.sessionId; + const res = await api.postTask(createdId, { input: [origin, ...input] }); + addSession(created.session); + // The remainder text has been carried into the new chat: discard the source session's input draft along with it. + discardSessionDraft(); + navigate(`/chat/${res.sessionId}`); + return true; + } catch (e) { + if (createdId) void api.deleteSession(createdId).catch(() => undefined); + toastError(apiErrorText(e, { modelId: ref.modelId })); + return false; + } + }, + [projectId, selected, addSession, discardSessionDraft, navigate], ); // @ handoff: doesn't use the current Session — creates a new chat for the @-mentioned agent @@ -510,9 +604,9 @@ export function ChatPage() { selected !== null && !stream.loading && !stream.error && stream.model.items.length === 0; // Input area in session state: Agent / Workspace / Model are already locked by the Session - // (the model selector isn't rendered; models is only used to look up the locked model's - // provider logo and display name for a read-only display) — only approval mode can still be - // changed (saved immediately on change). + // (the model selector isn't rendered; models feeds the locked model's read-only display and + // the /model switch picker) — approval mode and the per-turn thinking level stay editable; + // /model forks the conversation onto another model. const input = selected && ( ) isn't - * shown verbatim, it's collapsed into a single line reading "Handed off from 's chat"; - * when there's a source Session, the whole line is clickable and jumps back to the original chat - * (the source Session's title goes into the title hover tooltip, taking no space in the body). + * Provenance banners for conversations opened from another conversation — each collapses a + * machine-inserted source block (the raw text is never shown; the model still sees it): + * - `HandoffBanner` (``, @ delegation): "Handed off from 's chat"; + * - `ModelSwitchBanner` (``, the /model command): "switched model — + * continued from the earlier conversation". + * When there's a source Session, the whole line is clickable and jumps back to it (the + * source Session's title goes into the title hover tooltip, taking no space in the body). */ import { useNavigate } from "react-router"; import { S } from "../../lib/strings"; -import type { HandoffOrigin } from "./agent-mentions"; +import type { HandoffOrigin, ModelSwitchOrigin } from "./agent-mentions"; /** Display name of the source agent: `displayName (@id)` when the display name differs from the id, otherwise just `@id`. */ function agentLabel(origin: HandoffOrigin): string { @@ -15,20 +18,21 @@ function agentLabel(origin: HandoffOrigin): string { : `@${origin.agentId}`; } +const bannerFrame = + "anim-msg my-2 flex w-fit items-center gap-2 rounded-md border border-gray-200 bg-gray-50 px-3 py-2 text-xs text-gray-600 dark:border-gray-800 dark:bg-gray-900 dark:text-gray-400"; + export function HandoffBanner({ origin }: { origin: HandoffOrigin }) { const navigate = useNavigate(); const text = S.chat.handoffFrom(agentLabel(origin)); - const frame = - "anim-msg my-2 flex w-fit items-center gap-2 rounded-md border border-gray-200 bg-gray-50 px-3 py-2 text-xs text-gray-600 dark:border-gray-800 dark:bg-gray-900 dark:text-gray-400"; // A handoff initiated from draft state has no source Session: only the origin is shown, with nowhere to jump to. - if (!origin.sessionId) return

{text}

; + if (!origin.sessionId) return

{text}

; const sessionId = origin.sessionId; return ( ); } + +/** + * Notice for a conversation opened by the `/model` switch (`` first + * message): a single line naming the previous model, clickable to jump back to the source + * session — the same interaction as the handoff banner's back-link. + */ +export function ModelSwitchBanner({ origin }: { origin: ModelSwitchOrigin }) { + const navigate = useNavigate(); + const sessionId = origin.sessionId; + return ( + + ); +} diff --git a/packages/web/src/features/chat/message-item.tsx b/packages/web/src/features/chat/message-item.tsx index b09afc9..8f94961 100644 --- a/packages/web/src/features/chat/message-item.tsx +++ b/packages/web/src/features/chat/message-item.tsx @@ -19,10 +19,14 @@ import { ThinkingBlock } from "./thinking-block"; import { ToolCallCard } from "./tool-call-card"; import { SubagentCard } from "./subagent-card"; import { CompactionBanner } from "./compaction-banner"; -import { HandoffBanner } from "./handoff-banner"; +import { HandoffBanner, ModelSwitchBanner } from "./handoff-banner"; import { ScheduledBanner } from "./scheduled-banner"; import { SkillsBanner } from "./skills-banner"; -import { parseHandoffMessage, parseScheduledMessage } from "./agent-mentions"; +import { + parseHandoffMessage, + parseModelSwitchMessage, + parseScheduledMessage, +} from "./agent-mentions"; import { parseSkillsMessage } from "./skill-use"; import { TaskStatsLine } from "./task-stats-line"; import type { StreamRenderContext } from "./message-stream"; @@ -87,6 +91,9 @@ export function MessageItem({ item, ctx }: { item: ChatItem; ctx: StreamRenderCo // Source block for a chat created via @ handoff: collapsed into a single-line handoff notice (the raw text isn't shown), clickable to jump back to the original chat. const handoff = parseHandoffMessage(item.text); if (handoff) return ; + // Source block for a chat opened by the /model switch: collapsed into a single-line switch notice, clickable to jump back to the source conversation. + const modelSwitch = parseModelSwitchMessage(item.text); + if (modelSwitch) return ; // Source block for a scheduled-task trigger: collapsed into a single-line notice, with the task's prompt body rendered as usual (verbatim on the Trace page). const scheduled = parseScheduledMessage(item.text); // Source block for a skill invocation: parsing continues on scheduled's remaining body diff --git a/packages/web/src/features/chat/thinking-level.ts b/packages/web/src/features/chat/thinking-level.ts index 50bbf1c..fed079d 100644 --- a/packages/web/src/features/chat/thinking-level.ts +++ b/packages/web/src/features/chat/thinking-level.ts @@ -1,14 +1,22 @@ /** - * Pure logic for the conversation-time thinking-level picker (chat draft view). + * Pure logic for the conversation-time thinking-level pickers (chat draft + active session). * - * The picker is backed by the **Agent settings** (`system_config.model.thinking_level`): - * it shows the selected Agent's current level and writes a picked level straight through - * to the Agent config, so the session created on first send — which reads systemConfig - * fresh — runs with it, and it becomes the Agent's new default (switch-becomes-default). - * Per review: the menu lists the levels with short names only (no descriptions, no - * "default" row) under a title bar naming the control — and it does **not** offer "none" + * Two variants share these levels and labels, and both list only concrete levels — there is + * no "follow config" row anywhere: + * - Draft view: backed by the **Agent settings** (`system_config.model.thinking_level`) — + * shows the selected Agent's current level and writes a picked level straight through to + * the Agent config, so the session created on first send runs with it and it becomes the + * Agent's new default (switch-becomes-default). + * - Active session: the level is a **per-turn override**. The picker's displayed value + * initializes to the Agent config's level and auto-follows it while the user hasn't + * picked (internally the "" state remains as "untouched": tasks then omit the level, so + * the server/core fallback applies and config edits keep taking effect); an explicit pick + * sticks for the session and rides on every subsequent task as `thinkingLevel`, never + * writing through to the Agent config. + * Per review: the menus list the levels with short names only (no descriptions, no + * "default" row) under a title bar naming the control — and they do **not** offer "none" * (many models cannot disable thinking): "none" stays a valid stored/wire value, so a - * legacy config or session that carries it still displays via the label table below. + * legacy config or trace that carries it still displays via the label table below. */ /** All five stored/wire levels, for display lookup (mirrors core's ThinkingLevelName). */ diff --git a/packages/web/src/features/traces/trace-event-row.tsx b/packages/web/src/features/traces/trace-event-row.tsx index 2d2d18b..f90a995 100644 --- a/packages/web/src/features/traces/trace-event-row.tsx +++ b/packages/web/src/features/traces/trace-event-row.tsx @@ -162,7 +162,12 @@ function SessionMetaBody({ p }: { p: Record }) { ["source", String(p.source ?? "")], ["model_id", String(p.model_id ?? "")], ["context_window", String(p.model_context_window ?? "")], - ["thinking_level", String(p.thinking_level ?? "")], + // Legacy metas only: current traces no longer record a thinking level (it became a + // per-turn Task parameter), but old traces still carry the field — keep showing it + // for them (loose read) instead of losing the display. + ...(typeof p.thinking_level === "string" && p.thinking_level + ? ([["thinking_level", p.thinking_level]] as Array<[string, string]>) + : []), ["agent_state", String(p.agent_state ?? "")], ["workspace", String(p.workspace ?? "")], ]; diff --git a/packages/web/src/lib/omni/stream-model.ts b/packages/web/src/lib/omni/stream-model.ts index 7b7c57e..b7b25bc 100644 --- a/packages/web/src/lib/omni/stream-model.ts +++ b/packages/web/src/lib/omni/stream-model.ts @@ -241,13 +241,6 @@ export interface StreamModel { items: ChatItem[]; /** A nested sub-session model (produces no stats row; its stats count toward the parent). */ nested: boolean; - /** - * The session's thinking level, captured from the main session's `session_meta` on history - * replay ("default" when the Agent config leaves it unset; null until a session_meta has been - * seen). llmConfig is assembled once per session, so this is fixed for the session's lifetime — - * shown read-only in the input area next to the locked model. - */ - thinkingLevel: string | null; stats: TaskStatsTracker; /** The currently open text/thinking fragment (opened by start, closed by stop). */ openText: AssistantTextItem | null; @@ -325,7 +318,6 @@ function newModel(nested: boolean, localDecisions: Set): StreamModel { return { items: [], nested, - thinkingLevel: null, stats: createTaskStatsTracker(), openText: null, openThinking: null, @@ -400,13 +392,9 @@ export function pushMessage( advanceLastTs(model, msg.timestamp); return; } - // session_meta (main session): not rendered as an item, but its thinking level is captured - // for the input area's read-only display (fixed per session: llmConfig is assembled once at - // session creation, so a mid-conversation Agent-config change doesn't affect this session). - if (msg.type === "session_meta") { - const level = (msg.payload as { thinking_level?: unknown }).thinking_level; - if (typeof level === "string" && level) model.thinkingLevel = level; - } + // session_meta (main session): not rendered as an item and carries nothing the view model + // needs — session_meta holds per-session invariants (identity/config), all surfaced through + // the Session DTO instead. } /** ISO timestamp → milliseconds (returns undefined if invalid). */ diff --git a/packages/web/src/lib/strings-en.ts b/packages/web/src/lib/strings-en.ts index cc3901c..d083b8a 100644 --- a/packages/web/src/lib/strings-en.ts +++ b/packages/web/src/lib/strings-en.ts @@ -671,6 +671,13 @@ When done, open index.html in a browser and self-test once.`, handoffFrom: (agent: string) => `Handed off from ${agent}'s conversation`, handoffBack: (title?: string) => title ? `Back to the original conversation: ${title}` : "Back to the original conversation", + switchModel: "Switch model — continue this conversation in a new session", + switchModelTitle: "Switch model", + modelSwitchFrom: (prevModel?: string) => + prevModel + ? `Switched model (was ${prevModel}) — continued from the earlier conversation` + : "Switched model — continued from the earlier conversation", + modelSwitchAutoMessage: "Continue this conversation on the new model", scheduledFrom: (name: string) => `Triggered by scheduled task "${name}"`, emptyGreeting: "Start a new conversation", compactionRunning: (mode: string) => `Compaction in progress (${mode})…`, diff --git a/packages/web/src/lib/strings.ts b/packages/web/src/lib/strings.ts index 1bee768..2e97196 100644 --- a/packages/web/src/lib/strings.ts +++ b/packages/web/src/lib/strings.ts @@ -657,6 +657,13 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题( `使用 ${names.join("、")} 技能`, handoffFrom: (agent: string) => `由 ${agent} 的对话交接而来`, handoffBack: (title?: string) => (title ? `回到原对话:${title}` : "回到原对话"), + /** /model 切换:命令描述、拾取器标题、切换来源横幅与空正文自动消息。 */ + switchModel: "切换模型开启新会话延续本对话", + switchModelTitle: "切换模型", + modelSwitchFrom: (prevModel?: string) => + prevModel ? `已切换模型(原为 ${prevModel}),延续原会话` : "已切换模型,延续原会话", + /** /model 切换且正文为空时自动发送的首条消息正文(与 skillsAutoMessage 同一约定)。 */ + modelSwitchAutoMessage: "换用新模型继续这段对话", scheduledFrom: (name: string) => `由定时任务「${name}」触发`, emptyGreeting: "开始一段新对话", compactionRunning: (mode: string) => `压缩进行中(${mode})…`, diff --git a/packages/web/test/agent-mentions.test.ts b/packages/web/test/agent-mentions.test.ts index ad1bebb..1d4173c 100644 --- a/packages/web/test/agent-mentions.test.ts +++ b/packages/web/test/agent-mentions.test.ts @@ -1,8 +1,9 @@ /** * agent-mentions.ts unit tests: @ mention matching (cursor prefix / boundary * rules), candidate filtering, send-time parsing of a leading @ mention, - * generation of the first new-conversation origin block, and - * origin block parsing. + * generation of the first new-conversation origin block, + * origin block parsing, and the /model switch's + * origin block round-trip. */ import { describe, expect, it } from "vitest"; import type { AgentSummary } from "@prismshadow/penguin-server/api"; @@ -10,7 +11,9 @@ import { filterAgents, handoffMessage, matchMention, + modelSwitchMessage, parseHandoffMessage, + parseModelSwitchMessage, parseScheduledMessage, splitLeadingMention, } from "../src/features/chat/agent-mentions"; @@ -212,3 +215,37 @@ describe("parseScheduledMessage (parses the scheduled-task origin block, driving ).toBeNull(); }); }); + +describe("modelSwitchMessage / parseModelSwitchMessage (the /model switch origin block)", () => { + it("round-trips a full origin (session title with parentheses, trace path, workspace, previous pair)", () => { + const origin = { + sessionId: "session-2026-07-24-10-00-00-abcdef01", + sessionTitle: "Fix (the) parser", + tracePath: "/data/p/agents/a/traces/2026-07-24/session-x_001.jsonl", + workspace: "/data/ws", + prevProvider: "deepseek", + prevModelId: "deepseek-v4-pro", + }; + expect(parseModelSwitchMessage(modelSwitchMessage(origin))).toEqual(origin); + }); + + it("minimal origin (session id only) round-trips; optional lines are omitted from the block", () => { + const text = modelSwitchMessage({ sessionId: "session-01" }); + expect(text).not.toContain("trace:"); + expect(text).not.toContain("workspace:"); + expect(text).not.toContain("previous_model:"); + expect(parseModelSwitchMessage(text)).toEqual({ sessionId: "session-01" }); + }); + + it("plain messages and longer messages merely containing the block are not misdetected", () => { + expect(parseModelSwitchMessage("hello /model")).toBeNull(); + expect( + parseModelSwitchMessage(`before\n${modelSwitchMessage({ sessionId: "s1" })}`), + ).toBeNull(); + expect(parseModelSwitchMessage(`${modelSwitchMessage({ sessionId: "s1" })}\nafter`)).toBeNull(); + // A block without a session line is not an origin block. + expect( + parseModelSwitchMessage("\ntrace: /t.jsonl\n"), + ).toBeNull(); + }); +}); diff --git a/packages/web/test/stream-model.test.ts b/packages/web/test/stream-model.test.ts index ba6d668..382b03f 100644 --- a/packages/web/test/stream-model.test.ts +++ b/packages/web/test/stream-model.test.ts @@ -77,7 +77,6 @@ function meta(sessionId: string): OmniMessage { model_context_window: 200000, system_prompt: "", tools: [], - thinking_level: "default", agent_state: "/a", workspace: "/w", }); @@ -279,34 +278,6 @@ describe("approvals and events", () => { expect(items(m)).toHaveLength(0); }); - it("captures the session's thinking level from the main session_meta (read-only input-area tag)", () => { - const m = createStreamModel(); - expect(m.thinkingLevel).toBeNull(); - // The shared helper's meta carries thinking_level "default" (Agent config leaves it unset). - pushMessage(m, meta("session-x")); - expect(m.thinkingLevel).toBe("default"); - - const m2 = createStreamModel(); - pushMessage( - m2, - sessionMeta({ - session_id: "session-y", - model_id: "m", - provider: "custom", - model_context_window: 200000, - system_prompt: "", - tools: [], - thinking_level: "medium", - agent_state: "/a", - workspace: "/w", - }), - ); - expect(m2.thinkingLevel).toBe("medium"); - // An origin-tagged (sub-session) session_meta routes to the nested model and must not clobber the main session's level. - pushMessage(m2, withOrigin(meta("child"), "child")); - expect(m2.thinkingLevel).toBe("medium"); - }); - it("request_end final state timeout/malformed produces a retry notice item (with attempt number); request_begin marks it as resent", () => { const m = createStreamModel(); pushMessage(m, requestBegin()); diff --git a/packages/web/test/thinking-level.test.ts b/packages/web/test/thinking-level.test.ts index 742fd18..46bba61 100644 --- a/packages/web/test/thinking-level.test.ts +++ b/packages/web/test/thinking-level.test.ts @@ -1,8 +1,10 @@ /** - * thinking-level.ts unit tests: the conversation-time picker's short-name lookup and the - * selectable list — the picker offers only low/medium/high/xhigh (many models cannot disable - * thinking), while "none" stays a displayable stored value; "" (no override yet) and - * session_meta's "default" resolve to null (trigger shows a placeholder; the session tag hides). + * thinking-level.ts unit tests: the conversation-time pickers' short-name lookup and the + * selectable list — both pickers offer only low/medium/high/xhigh (many models cannot disable + * thinking; there is no "follow config" row — the session picker auto-follows the Agent + * config by initializing its display to it), while "none" stays a displayable stored value; + * "" (internally "untouched"/no override) and a legacy meta's "default" resolve to null so + * callers substitute the config level or a placeholder. */ import { describe, expect, it } from "vitest"; import { @@ -40,7 +42,7 @@ describe("thinking level lists", () => { }); describe("thinkingLevelLabel", () => { - it("returns null for non-levels: '' (no override yet), session_meta's 'default', unknown, null", () => { + it("returns null for non-levels: '' (untouched/no override), a legacy meta's 'default', unknown, null", () => { expect(thinkingLevelLabel(NAMES, "")).toBeNull(); expect(thinkingLevelLabel(NAMES, "default")).toBeNull(); expect(thinkingLevelLabel(NAMES, "ultra")).toBeNull();