diff --git a/packages/core/src/agent.ts b/packages/core/src/agent.ts index 72f9f9c..f58e25c 100644 --- a/packages/core/src/agent.ts +++ b/packages/core/src/agent.ts @@ -155,10 +155,11 @@ export function effectiveMaxContextLength(configured: number, contextWindow: unk * Output cap for meta requests (title generation / vision describing): these carry their own * small hardcoded budget, tightened further by the entry's per-model `max_tokens` when that is * smaller — a cap the user pinned below the budget must bind every request to that model. The - * budget is never raised. + * budget is never raised. A non-positive cap (-1 = uncapped) never tightens: Math.min against + * it would send max_tokens -1 on the wire, and meta requests must keep their small budget. */ export function metaMaxTokens(budget: number, modelCap: number | undefined): number { - return modelCap !== undefined ? Math.min(budget, modelCap) : budget; + return modelCap !== undefined && modelCap > 0 ? Math.min(budget, modelCap) : budget; } /** Create or load an Agent. */ diff --git a/packages/core/src/engine/context-engine.ts b/packages/core/src/engine/context-engine.ts index 6cef147..dbb4a77 100644 --- a/packages/core/src/engine/context-engine.ts +++ b/packages/core/src/engine/context-engine.ts @@ -125,7 +125,7 @@ export interface ContextEngineDeps { trace?: TraceSink; /** Engine initial state (derived by replaying Trace on Session resumption). */ initialState?: EngineInitialState; - /** Maximum LLM turns for a single Task. Defaults to 100. */ + /** Maximum LLM turns for a single Task. Defaults to 100; -1 removes the cap. */ maxTurns?: number; /** Maximum automatic retries for LLM timeout/reconnect within a single run. Defaults to 2. */ maxReconnects?: number; @@ -298,8 +298,11 @@ export class ContextEngine { let nextInput: OmniMessage[] = input; for (;;) { - // max_turns guard: emit a length notice and stop once exceeded. - if (turnCount >= this.maxTurns) { + // max_turns guard: emit a length notice and stop once exceeded. A non-positive cap + // (-1 per the config contract "must be > 0 or -1") disables the guard entirely — + // same convention as maxSessionTurns in shouldCompact (issue #55: -1 used to trip + // `0 >= -1` and stop before the first turn). + if (this.maxTurns > 0 && turnCount >= this.maxTurns) { // This turn's pending input (usually the previous turn's tool outputs) was never // submitted to the LLM: hold it as carry-over, to be resent merged with new input on // the next `run` (same as interruption-cleanup case A) — the previous turn's assistant diff --git a/packages/core/src/interfaces.ts b/packages/core/src/interfaces.ts index 486fa5c..d36e3c2 100644 --- a/packages/core/src/interfaces.ts +++ b/packages/core/src/interfaces.ts @@ -101,6 +101,7 @@ export interface GenerativeModelConfig { /** Full system Prompt after placeholder substitution in the system_config.system_prompt template. */ systemPrompt?: string; contextWindow?: number; + /** Output token cap per Request; non-positive (-1) means no explicit cap (omitted from the request). */ maxTokens?: number; thinkingLevel?: ThinkingLevelName; /** LLM Request timeout (ms): from system_config.model.timeoutMs; <=0 disables it. Defaults to 120000. */ diff --git a/packages/core/src/llm/generative-model.ts b/packages/core/src/llm/generative-model.ts index fa6661f..0751f0c 100644 --- a/packages/core/src/llm/generative-model.ts +++ b/packages/core/src/llm/generative-model.ts @@ -1113,7 +1113,10 @@ export function buildUniConfig(config: GenerativeModelConfig): UniConfig { if (config.systemPrompt !== undefined) { uniConfig.system_prompt = config.systemPrompt; } - if (config.maxTokens !== undefined) { + // Non-positive (-1 per the config contract) means "no explicit cap": the key is left off + // the wire so the provider default applies — sent literally, every provider rejects a + // negative max_tokens with a 400 (issue #55's sibling). + if (config.maxTokens !== undefined && config.maxTokens > 0) { uniConfig.max_tokens = config.maxTokens; } const thinking = mapThinkingLevel(config.thinkingLevel); diff --git a/packages/core/src/session.ts b/packages/core/src/session.ts index f273ac6..192ef24 100644 --- a/packages/core/src/session.ts +++ b/packages/core/src/session.ts @@ -38,6 +38,7 @@ export interface SessionConfig { llm: LLMInterface; environment: EnvironmentInterface; trace?: TraceSink; + /** Maximum LLM turns per Task (default 100; -1 removes the cap). */ maxTurns?: number; /** Creates a new LLM object after compaction (carries over the Session's accumulated Token count); context compaction is unavailable if not provided. */ createLLM?: (sessionTokens: TokenCounts) => LLMInterface; diff --git a/packages/core/test/agent.test.ts b/packages/core/test/agent.test.ts index 6852087..1310809 100644 --- a/packages/core/test/agent.test.ts +++ b/packages/core/test/agent.test.ts @@ -78,6 +78,7 @@ describe("metaMaxTokens (meta-request budget tightened by the per-model cap)", ( expect(metaMaxTokens(300, 8000)).toBe(300); // ample cap: the small budget stays expect(metaMaxTokens(300, 128)).toBe(128); // pinned below the budget: the cap binds expect(metaMaxTokens(2048, 1024)).toBe(1024); // vision-describer budget, same rule + expect(metaMaxTokens(300, -1)).toBe(300); // -1 = uncapped: never tightens to -1 (issue #55 sibling) }); }); diff --git a/packages/core/test/engine.test.ts b/packages/core/test/engine.test.ts index d54407f..dd7e712 100644 --- a/packages/core/test/engine.test.ts +++ b/packages/core/test/engine.test.ts @@ -331,6 +331,47 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => { ).toBe(true); }); + it("maxTurns -1 removes the cap instead of stopping before the first turn (issue #55)", async () => { + // Two tool-call turns followed by a final text turn: with the old `0 >= -1` guard the + // engine emitted the stop note without ever calling the LLM. + let calls = 0; + const llm: LLMInterface = { + async *streamGenerate() { + calls += 1; + if (calls <= 2) { + yield toolCall({ + name: "exec_command", + arguments: JSON.stringify({ cmd: "true" }), + toolCallId: `c${calls}`, + stopReason: "completed", + }); + } else { + yield assistantText("Done"); + } + yield tokenUsage(emptyTokenCounts(), { + cache_read: 0, + cache_write: 0, + output: 1, + total: 1, + }); + return { status: "completed" }; + }, + }; + const environment = new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }); + const engine = new ContextEngine({ llm, environment, maxTurns: -1 }); + + const all = await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(3); + const texts = all + .filter((m) => isCompleteModelMessage(m) && m.payload.type === "text") + .map((m) => (m.payload as TextPayload).text); + expect(texts.some((t) => t.includes("reached max turns"))).toBe(false); + expect(texts.some((t) => t === "Done")).toBe(true); + }); + it("max turns with pending tool outputs carries them over so the next run pairs the tool_call (issue #33)", async () => { const received: OmniMessage[][] = []; const llm: LLMInterface = { diff --git a/packages/core/test/llm.test.ts b/packages/core/test/llm.test.ts index ed3225a..a639221 100644 --- a/packages/core/test/llm.test.ts +++ b/packages/core/test/llm.test.ts @@ -1052,6 +1052,11 @@ describe("config helpers", () => { expect("system_prompt" in minimal).toBe(false); expect("max_tokens" in minimal).toBe(false); expect("thinking_level" in minimal).toBe(false); + + // max_tokens -1 (the config's "no cap" sentinel) stays OFF the wire — sent literally, + // providers reject a negative max_tokens with a 400. + const uncapped = buildUniConfig({ modelId: "m", tools: [], maxTokens: -1 }); + expect("max_tokens" in uncapped).toBe(false); }); it("omits tools when empty and never sets tool_choice (strict endpoints reject both)", () => { diff --git a/packages/docs/content/configuration.en.md b/packages/docs/content/configuration.en.md index 6089331..e3d2e3d 100644 --- a/packages/docs/content/configuration.en.md +++ b/packages/docs/content/configuration.en.md @@ -93,8 +93,8 @@ Edit this file via the CLI (`penguin config model …`) or the Web Models page | `description` | — | Agent description | | `version` | `1` | Agent State version (a natural number), incremented on each successful optimization | | `system_prompt` | built-in template | Required; the only template with placeholder substitution | -| `max_turns` | `100` | Maximum LLM turns per Task | -| `model.max_tokens` | `32000` | Output Token limit per Request | +| `max_turns` | `100` | Maximum LLM turns per Task (-1 removes the cap) | +| `model.max_tokens` | `32000` | Output Token limit per Request (-1 = no cap, provider default) | | `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh` | | `model.timeoutMs` | `120000` | Per-Request timeout (milliseconds) | | `compaction.max_context_length` | `128000` | Context Token threshold that triggers compaction | diff --git a/packages/docs/content/configuration.zh.md b/packages/docs/content/configuration.zh.md index aefea0f..15b8da7 100644 --- a/packages/docs/content/configuration.zh.md +++ b/packages/docs/content/configuration.zh.md @@ -93,8 +93,8 @@ output = 0.857143 | `description` | — | Agent 描述 | | `version` | `1` | Agent State 版本号(自然数),每次成功优化自增 | | `system_prompt` | 内置模板 | 必填;唯一进行占位符替换的模板 | -| `max_turns` | `100` | 单个 Task 的最大 LLM 轮数 | -| `model.max_tokens` | `32000` | 单次输出 Token 上限 | +| `max_turns` | `100` | 单个 Task 的最大 LLM 轮数(-1 不限制) | +| `model.max_tokens` | `32000` | 单次输出 Token 上限(-1 不设上限,用服务商默认) | | `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh` | | `model.timeoutMs` | `120000` | 单次 Request 超时(毫秒) | | `compaction.max_context_length` | `128000` | 触发压缩的上下文 Token 阈值 |