fix(core): honor the -1 sentinels for max_turns and max_tokens (#56)

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Yaowei Zheng
2026-07-24 16:18:38 +08:00
committed by GitHub
parent 6bbdac132f
commit c7465625a4
10 changed files with 66 additions and 10 deletions
+3 -2
View File
@@ -155,10 +155,11 @@ export function effectiveMaxContextLength(configured: number, contextWindow: unk
* Output cap for meta requests (title generation / vision describing): these carry their own
* small hardcoded budget, tightened further by the entry's per-model `max_tokens` when that is
* smaller — a cap the user pinned below the budget must bind every request to that model. The
* budget is never raised.
* budget is never raised. A non-positive cap (-1 = uncapped) never tightens: Math.min against
* it would send max_tokens -1 on the wire, and meta requests must keep their small budget.
*/
export function metaMaxTokens(budget: number, modelCap: number | undefined): number {
return modelCap !== undefined ? Math.min(budget, modelCap) : budget;
return modelCap !== undefined && modelCap > 0 ? Math.min(budget, modelCap) : budget;
}
/** Create or load an Agent. */
+6 -3
View File
@@ -125,7 +125,7 @@ export interface ContextEngineDeps {
trace?: TraceSink;
/** Engine initial state (derived by replaying Trace on Session resumption). */
initialState?: EngineInitialState;
/** Maximum LLM turns for a single Task. Defaults to 100. */
/** Maximum LLM turns for a single Task. Defaults to 100; -1 removes the cap. */
maxTurns?: number;
/** Maximum automatic retries for LLM timeout/reconnect within a single run. Defaults to 2. */
maxReconnects?: number;
@@ -298,8 +298,11 @@ export class ContextEngine {
let nextInput: OmniMessage[] = input;
for (;;) {
// max_turns guard: emit a length notice and stop once exceeded.
if (turnCount >= this.maxTurns) {
// max_turns guard: emit a length notice and stop once exceeded. A non-positive cap
// (-1 per the config contract "must be > 0 or -1") disables the guard entirely —
// same convention as maxSessionTurns in shouldCompact (issue #55: -1 used to trip
// `0 >= -1` and stop before the first turn).
if (this.maxTurns > 0 && turnCount >= this.maxTurns) {
// This turn's pending input (usually the previous turn's tool outputs) was never
// submitted to the LLM: hold it as carry-over, to be resent merged with new input on
// the next `run` (same as interruption-cleanup case A) — the previous turn's assistant
+1
View File
@@ -101,6 +101,7 @@ export interface GenerativeModelConfig {
/** Full system Prompt after placeholder substitution in the system_config.system_prompt template. */
systemPrompt?: string;
contextWindow?: number;
/** Output token cap per Request; non-positive (-1) means no explicit cap (omitted from the request). */
maxTokens?: number;
thinkingLevel?: ThinkingLevelName;
/** LLM Request timeout (ms): from system_config.model.timeoutMs; <=0 disables it. Defaults to 120000. */
+4 -1
View File
@@ -1113,7 +1113,10 @@ export function buildUniConfig(config: GenerativeModelConfig): UniConfig {
if (config.systemPrompt !== undefined) {
uniConfig.system_prompt = config.systemPrompt;
}
if (config.maxTokens !== undefined) {
// Non-positive (-1 per the config contract) means "no explicit cap": the key is left off
// the wire so the provider default applies — sent literally, every provider rejects a
// negative max_tokens with a 400 (issue #55's sibling).
if (config.maxTokens !== undefined && config.maxTokens > 0) {
uniConfig.max_tokens = config.maxTokens;
}
const thinking = mapThinkingLevel(config.thinkingLevel);
+1
View File
@@ -38,6 +38,7 @@ export interface SessionConfig {
llm: LLMInterface;
environment: EnvironmentInterface;
trace?: TraceSink;
/** Maximum LLM turns per Task (default 100; -1 removes the cap). */
maxTurns?: number;
/** Creates a new LLM object after compaction (carries over the Session's accumulated Token count); context compaction is unavailable if not provided. */
createLLM?: (sessionTokens: TokenCounts) => LLMInterface;