fix(core,docs): window-derived output caps and compaction threshold, slower retry ladder, retry-all-but-auth (#235)
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
+19
-17
@@ -30,7 +30,7 @@ import {
|
||||
type ModelRef,
|
||||
type ProjectConfig,
|
||||
} from "./state/index.js";
|
||||
import { GenerativeModel, ToolCallIdAllocator } from "./llm/index.js";
|
||||
import { GenerativeModel, ToolCallIdAllocator, effectiveMaxContextLength } from "./llm/index.js";
|
||||
import { Environment } from "./environment/index.js";
|
||||
import {
|
||||
Writer,
|
||||
@@ -132,18 +132,8 @@ export interface ResumeSessionOptions {
|
||||
baseUrl?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Effective compaction threshold: capped at 75% of the model's `context_window` —
|
||||
* the threshold must stay well below the hard window limit, otherwise small-window
|
||||
* models get rejected by the provider (a non-retryable 400) before compaction even
|
||||
* triggers, and the compaction request itself (old context + prompt + summary output)
|
||||
* also needs headroom. Not clamped when `<=0` (disabled) or the window is unknown.
|
||||
*/
|
||||
export function effectiveMaxContextLength(configured: number, contextWindow: unknown): number {
|
||||
if (configured <= 0) return configured;
|
||||
if (typeof contextWindow !== "number") return configured;
|
||||
return Math.min(configured, Math.floor(contextWindow * 0.75));
|
||||
}
|
||||
// Compaction-threshold derivation (`effectiveMaxContextLength`) lives with the rest of the
|
||||
// window arithmetic in llm/context-limits.ts; re-exported by llm/index.js.
|
||||
|
||||
/**
|
||||
* Output cap for meta requests (title generation / vision describing): these carry their own
|
||||
@@ -719,6 +709,11 @@ export class Agent {
|
||||
thinkingLevel: "none",
|
||||
// The describing budget, tightened by the vision entry's own pinned cap when smaller.
|
||||
maxTokens: metaMaxTokens(2048, visionEntry.max_tokens),
|
||||
// Same window derivation as ordinary requests (a formality here: the meta
|
||||
// budget is far below any real window, so the clamp never binds).
|
||||
...(visionEntry.context_window !== undefined
|
||||
? { contextWindow: visionEntry.context_window }
|
||||
: {}),
|
||||
requestTimeoutMs: 60_000,
|
||||
}),
|
||||
};
|
||||
@@ -749,10 +744,12 @@ export class Agent {
|
||||
});
|
||||
const tools = await environment.listTools();
|
||||
|
||||
// Effective output cap: the entry's per-model annotation wins over the Agent's
|
||||
// system_config value — the fit is a model trait: the seeded per-Agent default (32000)
|
||||
// cannot fit into e.g. a 32768-token context window together with any prompt, so a
|
||||
// small-window model needs its own pinned cap. Unset inherits the Agent value.
|
||||
// Configured output cap: the entry's per-model annotation wins over the Agent's
|
||||
// system_config value; unset inherits the Agent value. This is a ceiling, not the
|
||||
// literal wire value: GenerativeModel clamps each request's effective cap to what the
|
||||
// entry's context window can still fit (see llm/context-limits.ts, issue #218), so the
|
||||
// seeded per-Agent default (32000) no longer needs a manual per-model override to work
|
||||
// against e.g. a 32768-token window — setting the entry's `context_window` is enough.
|
||||
const maxTokens = modelEntry.max_tokens ?? this.state.systemConfig.model?.max_tokens;
|
||||
|
||||
// LLM constructor args are extracted into a constant so they can be reused as-is when
|
||||
@@ -801,6 +798,11 @@ export class Agent {
|
||||
thinkingLevel: "none",
|
||||
// The meta budget, tightened by the entry's pinned per-model cap when smaller.
|
||||
maxTokens: metaMaxTokens(300, modelEntry.max_tokens),
|
||||
// Same window derivation as ordinary requests (a formality here: the meta budget
|
||||
// is far below any real window, so the clamp never binds).
|
||||
...(modelEntry.context_window !== undefined
|
||||
? { contextWindow: modelEntry.context_window }
|
||||
: {}),
|
||||
requestTimeoutMs: 30_000,
|
||||
});
|
||||
|
||||
|
||||
@@ -169,15 +169,15 @@ export interface ContextEngineDeps {
|
||||
maxTurns?: number;
|
||||
/**
|
||||
* Maximum automatic retries for LLM timeout/reconnect within a single run. Defaults
|
||||
* to 5: with the default backoff (250ms base, 30s ceiling) that is 250+500+1000+2000+
|
||||
* 4000 ≈ 7.75s of total patience — the first attempts stay fast enough for transport
|
||||
* blips while the tail still gives transient provider rejections a window.
|
||||
* to 5: with the default backoff (2s base, 30s ceiling) that is 2+4+8+16+30 ≈ 60s of
|
||||
* total patience — transient provider failures (restarts, rate limits) get a real
|
||||
* recovery window instead of five retries burning out in about a second (issue #218).
|
||||
*/
|
||||
maxReconnects?: number;
|
||||
/**
|
||||
* Exponential backoff base (ms): the wait before reconnect retry N is
|
||||
* `base × 2^(N−1)`, capped at `reconnectBackoffMaxMs` (see reconnectDelayMs).
|
||||
* Defaults to 250.
|
||||
* Defaults to 2000.
|
||||
*/
|
||||
reconnectBackoffMs?: number;
|
||||
/** Ceiling (ms) for a single reconnect backoff wait. Defaults to 30000. */
|
||||
@@ -358,10 +358,14 @@ const RETRY_STATUSES: readonly StopReason[] = ["failed", "timeout", "malformed"]
|
||||
|
||||
/**
|
||||
* Delay before reconnect attempt N (1-based): exponential growth from `base` with a hard
|
||||
* ceiling `max` — `min(base × 2^(N−1), max)`. With the defaults (250ms base, 30s ceiling,
|
||||
* 5 reconnects) the ladder is 250, 500, 1000, 2000, 4000 ≈ 7.75s of total patience: one
|
||||
* shared schedule serves every retryable class, growing toward the slower ones (transient
|
||||
* provider quota errors) while the first steps stay as fast as a transport blip needs.
|
||||
* ceiling `max` — `min(base × 2^(N−1), max)`. With the defaults (2s base, 30s ceiling,
|
||||
* 5 reconnects) the ladder is 2s, 4s, 8s, 16s, 30s ≈ 60s of total patience: one shared
|
||||
* schedule serves every retryable class. The base is sized for the slow ones — transient
|
||||
* provider failures (restarts, rate limits) need seconds, not milliseconds, to
|
||||
* recover, and the old 250ms base burned the whole ladder in ~7.75s (issue #218); it also
|
||||
* keeps every planned wait at or above the hosts' 2s countdown floor (the Web App's
|
||||
* COUNTDOWN_MIN_MS), so no retry ever looks like a silent stall. Transport blips pay at
|
||||
* most one visible 2s wait — an acceptable trade for retries the user can see.
|
||||
*/
|
||||
export function reconnectDelayMs(base: number, max: number, attempt: number): number {
|
||||
return Math.min(base * 2 ** (attempt - 1), max);
|
||||
@@ -412,7 +416,7 @@ export class ContextEngine {
|
||||
constructor(private readonly deps: ContextEngineDeps) {
|
||||
this.maxTurns = deps.maxTurns ?? -1;
|
||||
this.maxReconnects = deps.maxReconnects ?? 5;
|
||||
this.reconnectBackoffMs = deps.reconnectBackoffMs ?? 250;
|
||||
this.reconnectBackoffMs = deps.reconnectBackoffMs ?? 2000;
|
||||
this.reconnectBackoffMaxMs = deps.reconnectBackoffMaxMs ?? 30_000;
|
||||
this.compactionMaxReconnects = deps.compactionMaxReconnects ?? this.maxReconnects;
|
||||
this.llm = deps.llm;
|
||||
|
||||
@@ -110,8 +110,19 @@ export interface GenerativeModelConfig {
|
||||
tools: ToolDefinition[];
|
||||
/** Full system Prompt after placeholder substitution in the system_config.system_prompt template. */
|
||||
systemPrompt?: string;
|
||||
/**
|
||||
* Model context window (tokens, from the model entry). Used to clamp each request's
|
||||
* effective output cap so `input + max_tokens` stays inside the window (issue #218).
|
||||
* Unset (or implausibly small, see llm/context-limits.ts resolveContextWindow): the
|
||||
* clamp is disabled — a hard cap is never derived from an assumed window.
|
||||
*/
|
||||
contextWindow?: number;
|
||||
/** Output token cap per Request; non-positive (-1) means no explicit cap (omitted from the request). */
|
||||
/**
|
||||
* Output token cap per Request; non-positive (-1) means no explicit cap (omitted from the
|
||||
* request). With `contextWindow` set, a positive cap is a ceiling, not a constant: each
|
||||
* request sends `min(maxTokens, contextWindow − estimated input − safety margin)` (see
|
||||
* llm/context-limits.ts) so small-window models never fail provider validation.
|
||||
*/
|
||||
maxTokens?: number;
|
||||
/** Construction-time default thinking level; a per-request `GenerativeModelParameters.thinkingLevel` overrides it for that request. */
|
||||
thinkingLevel?: ThinkingLevelName;
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
/**
|
||||
* Window-derived request limits (issue #218) — pure arithmetic shared by the per-request
|
||||
* output-token clamp (GenerativeModel) and the compaction-threshold derivation (Agent).
|
||||
*
|
||||
* The problem both solve: a model entry's `max_tokens` and the Agent's
|
||||
* `compaction.max_context_length` are fixed numbers, while the space they must fit into is
|
||||
* the model's `context_window` minus whatever the input already occupies. Small-window
|
||||
* models (a local vLLM with `--max-model-len 32768`, say) reject any request whose
|
||||
* `input + max_tokens` exceeds the window with a non-retryable 400 — with the seeded
|
||||
* per-Agent `max_tokens` of 32000 that is *every* request, before compaction ever gets a
|
||||
* chance to run. Everything here is derived at use from the configured values; stored
|
||||
* config is never rewritten.
|
||||
*
|
||||
* `OUTPUT_SAFETY_MARGIN` is the single tunable of the margin story: the output floor and
|
||||
* the compaction headroom are derived from it (PR #235 review), so their invariants —
|
||||
* floor below margin, headroom above margin — cannot drift apart. The two remaining
|
||||
* independent facts are `DEFAULT_CONTEXT_WINDOW` and `IMAGE_TOKEN_ESTIMATE`.
|
||||
*
|
||||
* Thinking-level interaction (considered, out of scope): on a high-reasoning session a
|
||||
* small derived cap can be consumed by thinking tokens before any visible answer, ending
|
||||
* the request as length -> failed. The arithmetic has no notion of the thinking level;
|
||||
* the compaction threshold firing `COMPACTION_HEADROOM` below the window keeps caps from
|
||||
* staying small for long, which bounds the exposure.
|
||||
*/
|
||||
import type { OmniMessage } from "../omnimessage/index.js";
|
||||
|
||||
/**
|
||||
* Assumed context window when a model entry has no usable `context_window` configured.
|
||||
* Mirrors the web app's display default (`packages/web/src/lib/context.ts`,
|
||||
* `DEFAULT_CONTEXT_WINDOW`) so every consumer of an unknown window reasons from the same
|
||||
* number. Only the compaction-threshold derivation uses this assumption; the per-request
|
||||
* output clamp is disabled without a configured window (see effectiveMaxOutputTokens).
|
||||
* Models with a *smaller* real window still need `context_window` set on their entry —
|
||||
* no derivation can protect a window it doesn't know about.
|
||||
*/
|
||||
export const DEFAULT_CONTEXT_WINDOW = 128000;
|
||||
|
||||
/**
|
||||
* Smallest `context_window` value taken at face value. Anything below is treated as
|
||||
* unconfigured (no real model has a window under 4096; such values are typos or unit
|
||||
* mistakes), which routes the derivations to their unconfigured behavior instead of
|
||||
* clamping every request into uselessness against a bogus number.
|
||||
*/
|
||||
export const MIN_USABLE_CONTEXT_WINDOW = 4096;
|
||||
|
||||
/**
|
||||
* Safety margin (tokens) kept between `estimated input + output cap` and the window.
|
||||
* It absorbs what the estimate cannot see: provider-side chat-template and tool-schema
|
||||
* serialization overhead, and the error of the character heuristic on input appended since
|
||||
* the last real `token_usage`. 1024 comfortably covers those sources while staying
|
||||
* negligible against every real window (>= 8k).
|
||||
*/
|
||||
export const OUTPUT_SAFETY_MARGIN = 1024;
|
||||
|
||||
/**
|
||||
* Floor for the derived output cap: the minimal useful output budget. The clamp never
|
||||
* emits a non-positive (or uselessly tiny) `max_tokens` — when the remaining window is
|
||||
* smaller than this, compaction should already have fired (its threshold sits
|
||||
* `COMPACTION_HEADROOM` below the window); if it hasn't (compaction disabled, or a
|
||||
* misconfigured window), a deterministic small cap is still better than sending a
|
||||
* negative/zero cap the provider rejects outright. Derived at half the margin so a
|
||||
* floored request still fits inside the window when the input estimate is accurate.
|
||||
*/
|
||||
export const MIN_OUTPUT_TOKENS = Math.floor(OUTPUT_SAFETY_MARGIN / 2);
|
||||
|
||||
/**
|
||||
* Headroom (tokens) reserved under the model window when deriving the effective compaction
|
||||
* threshold: one `OUTPUT_SAFETY_MARGIN` for the margin itself plus the same again (~1k)
|
||||
* for the summary request's own output — the compaction request runs through the same
|
||||
* per-request clamp, so at the trigger point its output budget is about
|
||||
* `COMPACTION_HEADROOM − OUTPUT_SAFETY_MARGIN` minus the compaction prompt; reserving less
|
||||
* would clamp the summary to the floor and truncate it. On a 32768 window this makes
|
||||
* compaction fire at ≈ 30720 instead of never (the previous default threshold of 128000
|
||||
* was unreachable inside the window).
|
||||
*/
|
||||
export const COMPACTION_HEADROOM = OUTPUT_SAFETY_MARGIN * 2;
|
||||
|
||||
/**
|
||||
* Flat allowance for one image input: provider vision token counts vary widely
|
||||
* (~100–2000+); counting a data URI's base64 characters instead would overestimate a
|
||||
* hundredfold.
|
||||
*/
|
||||
const IMAGE_TOKEN_ESTIMATE = 1600;
|
||||
|
||||
/**
|
||||
* Resolves a model entry's `context_window` to a usable number: a finite number at or
|
||||
* above {@link MIN_USABLE_CONTEXT_WINDOW} wins; anything else — unset, non-numeric,
|
||||
* non-positive, or implausibly small — is `undefined` ("unconfigured"). Callers decide
|
||||
* what unconfigured means for them: the output clamp switches itself off, the compaction
|
||||
* derivation falls back to {@link DEFAULT_CONTEXT_WINDOW}.
|
||||
*/
|
||||
export function resolveContextWindow(contextWindow: unknown): number | undefined {
|
||||
return typeof contextWindow === "number" &&
|
||||
Number.isFinite(contextWindow) &&
|
||||
contextWindow >= MIN_USABLE_CONTEXT_WINDOW
|
||||
? contextWindow
|
||||
: undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Crude token estimate for a text: ASCII at ~4 characters per token, everything else
|
||||
* (CJK etc.) at 1 token per character. Deliberately a character heuristic, not a
|
||||
* tokenizer — it only ever feeds derivations that keep {@link OUTPUT_SAFETY_MARGIN} in
|
||||
* reserve, and the non-ASCII bucket errs high (safe direction: a high input estimate
|
||||
* shrinks the output cap, a low one risks a provider 400).
|
||||
*/
|
||||
export function approximateTokens(text: string): number {
|
||||
let ascii = 0;
|
||||
let wide = 0;
|
||||
for (const ch of text) {
|
||||
if ((ch.codePointAt(0) ?? 0) < 0x80) ascii += 1;
|
||||
else wide += 1;
|
||||
}
|
||||
return Math.ceil(ascii / 4) + wide;
|
||||
}
|
||||
|
||||
/**
|
||||
* Token estimate for a batch of input messages: image content gets a flat allowance,
|
||||
* everything else is estimated from its serialized payload (covering user text, tool
|
||||
* outputs and tool calls uniformly; the JSON syntax counted along the way is a small
|
||||
* overestimate — the safe direction, and it also stands in for chat-template structure).
|
||||
*
|
||||
* Images appear in two shapes and both must bypass the character count: as a payload of
|
||||
* their own (`image_url` / `inline_data`), and as the `images` array of data URLs riding
|
||||
* on a tool output (complete or partial — `read_image` and screenshot-returning tools).
|
||||
* Serializing those data URLs would count every base64 character: a 1 MB image would
|
||||
* estimate ≈ 262k "tokens" (~163x over) and floor the next request's output cap even on a
|
||||
* 128k window (PR #235 review).
|
||||
*/
|
||||
export function approximateMessagesTokens(messages: OmniMessage[]): number {
|
||||
let total = 0;
|
||||
for (const msg of messages) {
|
||||
const p = msg.payload as { type?: string; images?: unknown };
|
||||
if (p.type === "image_url" || p.type === "inline_data") {
|
||||
total += IMAGE_TOKEN_ESTIMATE;
|
||||
} else if (Array.isArray(p.images)) {
|
||||
const { images, ...rest } = p as { images: unknown[] };
|
||||
total += images.length * IMAGE_TOKEN_ESTIMATE + approximateTokens(JSON.stringify(rest));
|
||||
} else {
|
||||
total += approximateTokens(JSON.stringify(msg.payload));
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/**
|
||||
* Effective output-token cap for one request:
|
||||
* `min(configured max_tokens, max(context_window − estimated input − OUTPUT_SAFETY_MARGIN,
|
||||
* MIN_OUTPUT_TOKENS))` — floor the remaining window, then never exceed the configured cap,
|
||||
* so the clamp only ever *lowers* a cap (a configured cap already below the floor, e.g. a
|
||||
* pinned meta budget, comes back verbatim).
|
||||
*
|
||||
* `undefined` when no positive cap is configured — the "no explicit cap" contract (`-1` /
|
||||
* unset) is preserved: the key stays off the wire and the provider's own default applies
|
||||
* (OpenAI-compatible servers, vLLM included, then bound the output by the remaining window
|
||||
* themselves). With no usable `contextWindow` the configured cap is returned unchanged:
|
||||
* a hard per-request clamp must not be derived from an assumption — a large-window model
|
||||
* whose entry simply omits `context_window` would otherwise get its outputs floored past
|
||||
* the assumed mark when compaction is disabled (PR #235 review). For big-window models the
|
||||
* subtraction stays far above the configured cap, so `min` returns it unchanged — a no-op
|
||||
* in practice.
|
||||
*/
|
||||
export function effectiveMaxOutputTokens(
|
||||
configured: number | undefined,
|
||||
contextWindow: number | undefined,
|
||||
estimatedInputTokens: number,
|
||||
): number | undefined {
|
||||
if (configured === undefined || configured <= 0) return undefined;
|
||||
if (contextWindow === undefined) return configured;
|
||||
const remaining = contextWindow - estimatedInputTokens - OUTPUT_SAFETY_MARGIN;
|
||||
return Math.min(configured, Math.max(remaining, MIN_OUTPUT_TOKENS));
|
||||
}
|
||||
|
||||
/**
|
||||
* Effective compaction threshold: the configured `compaction.max_context_length` capped at
|
||||
* `context_window − COMPACTION_HEADROOM` — the threshold must stay below the hard window
|
||||
* limit by enough to fit the compaction request itself (prompt + summary output + safety
|
||||
* margin), otherwise small-window models get rejected by the provider (a non-retryable
|
||||
* 400) before compaction even triggers. An unconfigured window (unset, or below
|
||||
* {@link MIN_USABLE_CONTEXT_WINDOW}) derives from {@link DEFAULT_CONTEXT_WINDOW}: unlike
|
||||
* the output clamp, a threshold derived from the assumption is harmless-to-helpful — it
|
||||
* only makes compaction fire by the assumed mark. Not clamped when `<=0` (compaction
|
||||
* disabled — the value keeps its "off" meaning); with every usable window at least
|
||||
* `MIN_USABLE_CONTEXT_WINDOW`, the derived threshold is always positive and can never
|
||||
* silently flip into the "disabled" contract.
|
||||
*/
|
||||
export function effectiveMaxContextLength(configured: number, contextWindow: unknown): number {
|
||||
if (configured <= 0) return configured;
|
||||
const window = resolveContextWindow(contextWindow) ?? DEFAULT_CONTEXT_WINDOW;
|
||||
return Math.min(configured, window - COMPACTION_HEADROOM);
|
||||
}
|
||||
@@ -13,13 +13,13 @@
|
||||
* 3. Interruption/error handling: `finishInterrupted` first closes any open
|
||||
* streaming segments and backfills the complete message, then the output ends — never
|
||||
* leaking a malformed structure. This interface **never retries internally** — it only
|
||||
* classifies, and `context_engine` owns the retry policy: errors recognised as retryable
|
||||
* (network/transport drops, timeouts, 429/5xx, provider quota exhaustion, see
|
||||
* `isRetryableError`) end with `timeout`; AgentHub JSON parse errors end with `malformed`;
|
||||
* everything else the classifier does not recognise ends with `failed` — which the engine
|
||||
* reconnects on all the same, because this classifier is an allowlist and a gateway
|
||||
* wording a transient fault its own way falls through it. User interruption ends with
|
||||
* `aborted`; credentials failures end with their own terminal status `auth` (see
|
||||
* labels, and `context_engine` owns the retry policy: every LLM error retries on the
|
||||
* engine's ladder EXCEPT `auth`. The label picks the taxonomy, not the policy:
|
||||
* transport-shaped errors (network/transport drops, timeouts, 429/5xx, see
|
||||
* `isRetryableError`) end with `timeout`; AgentHub JSON parse errors end with
|
||||
* `malformed`; everything else — provider 4xx rejections included — ends with
|
||||
* `failed`, and the engine reconnects on all three the same. User interruption ends
|
||||
* with `aborted`; credentials failures end with their own terminal status `auth` (see
|
||||
* `isAuthenticationError`) — the one status the engine refuses to retry, and the one
|
||||
* hosts key on to gate input.
|
||||
*
|
||||
@@ -69,6 +69,12 @@ import type {
|
||||
ToolDefinition,
|
||||
} from "../interfaces.js";
|
||||
import { ToolCallIdAllocator, stripToolCallIdSuffix } from "./tool-call-ids.js";
|
||||
import {
|
||||
approximateMessagesTokens,
|
||||
approximateTokens,
|
||||
effectiveMaxOutputTokens,
|
||||
resolveContextWindow,
|
||||
} from "./context-limits.js";
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pure conversion function: OmniMessage[] → a single UniMessage (unit-testable, no network)
|
||||
@@ -821,9 +827,6 @@ const RETRYABLE_NETWORK_CODES: ReadonlySet<string> = new Set([
|
||||
"UND_ERR_BODY_TIMEOUT",
|
||||
]);
|
||||
|
||||
/** Provider "quota / subscription exhausted" error codes (OpenAI-compatible bodies). */
|
||||
const QUOTA_CODES: ReadonlySet<string> = new Set(["insufficient_user_quota", "insufficient_quota"]);
|
||||
|
||||
/** Credentials/authentication error codes and types (OpenAI-compatible bodies / SDK errors). */
|
||||
const AUTH_CODES: ReadonlySet<string> = new Set([
|
||||
"invalid_api_key",
|
||||
@@ -848,13 +851,13 @@ function anyInCauseChain(error: unknown, probe: (level: object) => boolean): boo
|
||||
}
|
||||
|
||||
/**
|
||||
* Provider error-code signals at one level of an error: `code` on the error itself (the
|
||||
* OpenAI SDK exposes the parsed body's code directly), plus the parsed body under `error` —
|
||||
* both the OpenAI shape (`err.error.code`) and the Anthropic SDK shape, whose `error`
|
||||
* property holds the whole response body (`err.error.error.code`). With `includeType`, the
|
||||
* matching `type` fields are collected too (auth signals ride on `type` in some bodies).
|
||||
* Provider error-code signals at one level of an error: `code` and `type` on the error
|
||||
* itself (the OpenAI SDK exposes the parsed body's code directly), plus the parsed body
|
||||
* under `error` — both the OpenAI shape (`err.error.code`) and the Anthropic SDK shape,
|
||||
* whose `error` property holds the whole response body (`err.error.error.code`). Auth
|
||||
* signals ride on `type` in some bodies, so both fields are collected.
|
||||
*/
|
||||
function providerSignals(level: object, includeType: boolean): string[] {
|
||||
function providerSignals(level: object): string[] {
|
||||
const err = level as {
|
||||
code?: unknown;
|
||||
type?: unknown;
|
||||
@@ -862,45 +865,10 @@ function providerSignals(level: object, includeType: boolean): string[] {
|
||||
};
|
||||
const body = typeof err.error === "object" && err.error !== null ? err.error : undefined;
|
||||
const inner = typeof body?.error === "object" && body.error !== null ? body.error : undefined;
|
||||
const signals = [err.code, body?.code, inner?.code];
|
||||
if (includeType) signals.push(err.type, body?.type, inner?.type);
|
||||
const signals = [err.code, body?.code, inner?.code, err.type, body?.type, inner?.type];
|
||||
return signals.filter((v): v is string => typeof v === "string");
|
||||
}
|
||||
|
||||
/**
|
||||
* Determines whether an error is the provider's "quota / subscription exhausted" rejection
|
||||
* (e.g. a gateway 403 whose body carries `insufficient_user_quota`). Topping up or the
|
||||
* billing cycle rolling over fixes it without touching the Session, so it is transient and
|
||||
* goes through the engine's reconnect flow like a network problem. Deliberately tight:
|
||||
* only a payment/permission status (402/403 — or no status at all) combined with a known
|
||||
* quota code (own, `cause` chain, or parsed provider body), or a 403 whose message names
|
||||
* quota/subscription; a plain 403 (permission denied) stays non-retryable.
|
||||
*/
|
||||
export function isQuotaExhaustedError(error: unknown): boolean {
|
||||
if (error == null || typeof error !== "object") return false;
|
||||
// Belt to the auth-first ordering in isRetryableError: an error carrying a definitive
|
||||
// auth signal is never quota, no matter what its message says — a 403 whose body codes
|
||||
// `invalid_api_key` but whose message mentions a subscription must not classify as
|
||||
// retryable-quota and burn through the reconnect ladder against a dead credential.
|
||||
if (isAuthenticationError(error)) return false;
|
||||
const err = error as { status?: number; statusCode?: number; message?: unknown };
|
||||
const status = err.status ?? err.statusCode;
|
||||
// Any other status keeps its own classification (401 auth, 429 rate limit, …).
|
||||
if (typeof status === "number" && status !== 402 && status !== 403) return false;
|
||||
if (
|
||||
anyInCauseChain(error, (level) => providerSignals(level, false).some((c) => QUOTA_CODES.has(c)))
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
// 额度 covers Chinese-language gateway messages such as 订阅额度不足 ("subscription quota
|
||||
// exhausted") that carry no machine-readable code.
|
||||
return (
|
||||
status === 403 &&
|
||||
typeof err.message === "string" &&
|
||||
/quota|subscription|额度/i.test(err.message)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Determines whether an error is a credentials/authentication failure — the one class an
|
||||
* in-run retry can never fix: the request keeps going out with the same dead credential.
|
||||
@@ -909,41 +877,46 @@ export function isQuotaExhaustedError(error: unknown): boolean {
|
||||
* API key (Models page) — after which the Session can continue — not retrying. Signals:
|
||||
* HTTP 401 (any), a known auth code/type (on the error, its `cause` chain, or the parsed
|
||||
* provider body), or the SDK error class name `AuthenticationError` (OpenAI / Anthropic
|
||||
* SDKs). A bare 403 carries no such signal and stays a plain failure (usually
|
||||
* permission/policy, not credentials).
|
||||
* SDKs). Deliberately narrow and explicit — this detector is the ONLY thing that stops a
|
||||
* request from retrying, so nothing heuristic belongs in it: a bare 403 carries no
|
||||
* credential signal, classifies like any other failure, and rides the retry ladder.
|
||||
*/
|
||||
export function isAuthenticationError(error: unknown): boolean {
|
||||
return anyInCauseChain(error, (level) => {
|
||||
const err = level as { status?: unknown; statusCode?: unknown; name?: unknown };
|
||||
if (err.status === 401 || err.statusCode === 401) return true;
|
||||
if (providerSignals(level, true).some((c) => AUTH_CODES.has(c))) return true;
|
||||
if (providerSignals(level).some((c) => AUTH_CODES.has(c))) return true;
|
||||
const name = typeof err.name === "string" ? err.name : "";
|
||||
return name === "AuthenticationError" || level.constructor?.name === "AuthenticationError";
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Determines whether an error is retryable.
|
||||
* Determines whether an error is transport-shaped — the `timeout` vs `failed` LABEL for
|
||||
* an errored request, not the retry gate: the engine retries every LLM error except
|
||||
* `auth`, whatever this returns. What the label buys is honest taxonomy for
|
||||
* observability and hosts — `timeout` reads as "connectivity/provider hiccup", `failed`
|
||||
* as "the provider rejected the request" — while both ride the same reconnect ladder.
|
||||
*
|
||||
* Retryable: network/transport errors (including undici's `UND_ERR_*` on the `cause` chain),
|
||||
* timeouts, connection reset, HTTP 429 / 5xx, and provider quota/subscription exhaustion
|
||||
* (see `isQuotaExhaustedError`).
|
||||
* Not retryable: authentication errors (checked FIRST — a definitive credential signal is
|
||||
* never reclassified by the heuristics below, see `isAuthenticationError`) and other HTTP
|
||||
* 4xx auth/parameter errors (401/403/400/404, etc.).
|
||||
* JSON parse errors are classified separately as `malformed` by `isMalformedJsonParseError`.
|
||||
* `timeout`-shaped: network/transport errors (including undici's `UND_ERR_*` on the
|
||||
* `cause` chain), timeouts, connection reset, HTTP 429 / 5xx.
|
||||
* `failed`-shaped: everything else, HTTP 4xx provider rejections included (a quota or
|
||||
* subscription rejection lands here too — it retries all the same and its real message
|
||||
* rides on the outcome).
|
||||
* Authentication errors are checked FIRST and are never transport-shaped (see
|
||||
* `isAuthenticationError`); JSON parse errors are classified separately as `malformed`
|
||||
* by `isMalformedJsonParseError`.
|
||||
*
|
||||
* Since AgentHub doesn't guarantee the shape of error objects, this uses a lenient check: first
|
||||
* the status code, then error codes / message keywords. When undeterminable, treat it as
|
||||
* **non-retryable** to avoid pointless retries.
|
||||
* Since AgentHub doesn't guarantee the shape of error objects, this uses a lenient check:
|
||||
* first the status code, then error codes / message keywords. When undeterminable, label
|
||||
* it `failed` — still retried, just reported for what it is.
|
||||
*/
|
||||
export function isRetryableError(error: unknown): boolean {
|
||||
if (error == null) return false;
|
||||
|
||||
// 0. Authentication is definitive and terminal: retrying a dead credential can never
|
||||
// succeed, and no heuristic below (quota keywords, message vocabulary) may reclassify
|
||||
// it — this ordering is the only thing keeping a dead credential out of the retry
|
||||
// ladder now that quota errors are deliberately retryable.
|
||||
// succeed, and no vocabulary heuristic below may reclassify it — a definitive
|
||||
// credential signal always wins.
|
||||
if (isAuthenticationError(error)) return false;
|
||||
|
||||
const err = error as {
|
||||
@@ -958,10 +931,7 @@ export function isRetryableError(error: unknown): boolean {
|
||||
if (typeof status === "number") {
|
||||
if (status === 429 || status === 408) return true; // Rate limited / request timeout (transient)
|
||||
if (status >= 500 && status <= 599) return true; // Server error
|
||||
// Provider quota/subscription exhaustion (402/403 + explicit signals): transient —
|
||||
// checked before the blanket 4xx rule below would discard it.
|
||||
if (isQuotaExhaustedError(error)) return true;
|
||||
if (status >= 400 && status <= 499) return false; // Other 4xx auth/parameter errors, not retryable
|
||||
if (status >= 400 && status <= 499) return false; // Provider rejection: labeled failed (still retried by the engine)
|
||||
}
|
||||
|
||||
// 2. Network/transport error codes, probing the `cause` chain (Node fetch wraps the real
|
||||
@@ -975,9 +945,6 @@ export function isRetryableError(error: unknown): boolean {
|
||||
return true;
|
||||
}
|
||||
|
||||
// 2b. Status-less quota signals (some gateways surface only the provider code).
|
||||
if (isQuotaExhaustedError(error)) return true;
|
||||
|
||||
// 3. Error name / message keywords (timeout, network, rate limit, undici disconnects).
|
||||
if (err.name === "AbortError") return false; // User interruption, not retryable
|
||||
const text = `${err.name ?? ""} ${err.message ?? ""}`.toLowerCase();
|
||||
@@ -1028,6 +995,28 @@ export class GenerativeModel implements LLMInterface {
|
||||
* instance rebuilt on compaction; defaults to a fresh one.
|
||||
*/
|
||||
private readonly toolCallIds: ToolCallIdAllocator;
|
||||
/** Configured output cap (`GenerativeModelConfig.maxTokens`); the per-request clamp derives the effective cap from it (see effectiveMaxTokens). */
|
||||
private readonly configuredMaxTokens: number | undefined;
|
||||
/** Model context window; `undefined` when unconfigured (or implausibly small, see resolveContextWindow) — the per-request clamp then disables itself rather than clamp against an assumption. */
|
||||
private readonly contextWindow: number | undefined;
|
||||
/**
|
||||
* Construction-time estimate of the fixed request prefix (system prompt + tool schemas):
|
||||
* the prefix is part of every request but never part of `newMessages`. Seeds
|
||||
* `lastRequestTotal`, and re-seeds it in `setHistory`.
|
||||
*/
|
||||
private readonly baseInputTokens: number;
|
||||
/**
|
||||
* The current context's size: the most recent completed request's real
|
||||
* `token_usage.request.total` once one exists (a measured total always includes the
|
||||
* prefix, so it can only refine the seed upward from real data; providers stripping
|
||||
* historical thinking only make it an overestimate, the safe direction), the
|
||||
* `baseInputTokens` seed before that, plus the replayed-history estimate after
|
||||
* `setHistory`. The next request's input is this figure plus the newly appended
|
||||
* messages.
|
||||
*/
|
||||
private lastRequestTotal: number;
|
||||
/** Last hard-clamped cap already warned about on stderr (dedupe: retries reuse the same estimate and would repeat the identical line). */
|
||||
private lastWarnedCap: number | undefined;
|
||||
|
||||
/** Cumulative session tokens. */
|
||||
sessionTokens: TokenCounts = emptyTokenCounts();
|
||||
@@ -1048,18 +1037,78 @@ export class GenerativeModel implements LLMInterface {
|
||||
this.defaultThinkingLevel = config.thinkingLevel;
|
||||
this.requestTimeoutMs = config.requestTimeoutMs ?? 120000;
|
||||
this.toolCallIds = config.toolCallIds ?? new ToolCallIdAllocator();
|
||||
this.configuredMaxTokens = config.maxTokens;
|
||||
this.contextWindow = resolveContextWindow(config.contextWindow);
|
||||
this.baseInputTokens =
|
||||
approximateTokens(config.systemPrompt ?? "") +
|
||||
approximateTokens(JSON.stringify(this.uniConfig.tools ?? []));
|
||||
this.lastRequestTotal = this.baseInputTokens;
|
||||
}
|
||||
|
||||
/**
|
||||
* The UniConfig for one request: the shared frozen config plus this request's effective
|
||||
* thinking level (per-request override, else the construction-time default; neither → the
|
||||
* key stays off the wire, preserving the provider default).
|
||||
* key stays off the wire, preserving the provider default) and this request's effective
|
||||
* output cap (the window-derived clamp below; equal to the configured cap for big-window
|
||||
* models, so the frozen value goes out unchanged).
|
||||
*/
|
||||
private requestConfig(override: ThinkingLevelName | undefined): UniConfig {
|
||||
private requestConfig(
|
||||
override: ThinkingLevelName | undefined,
|
||||
newMessages: OmniMessage[],
|
||||
): UniConfig {
|
||||
const thinking = mapThinkingLevel(override ?? this.defaultThinkingLevel);
|
||||
return thinking === undefined
|
||||
? this.uniConfig
|
||||
: { ...this.uniConfig, thinking_level: thinking };
|
||||
const maxTokens = this.effectiveMaxTokens(newMessages);
|
||||
let cfg = this.uniConfig;
|
||||
if (thinking !== undefined) cfg = { ...cfg, thinking_level: thinking };
|
||||
if (maxTokens !== undefined && maxTokens !== this.uniConfig.max_tokens) {
|
||||
cfg = { ...cfg, max_tokens: maxTokens };
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-request output cap: `min(configured max_tokens, context_window − estimated input −
|
||||
* safety margin)`, floored — recomputed for every request (compaction requests included:
|
||||
* they run through the same path, exactly when the context is largest) from the freshest
|
||||
* input knowledge this object has: the last completed request's real `token_usage` total
|
||||
* plus a character-heuristic estimate of the newly appended messages (no tokenizer;
|
||||
* see context-limits.ts). Fixes issue #218: a fixed cap (the seeded 32000) that ignores
|
||||
* the input made every request to a small-window model (e.g. a 32k vLLM) fail provider
|
||||
* validation with a non-retryable 400.
|
||||
*
|
||||
* Interplay with compaction: the engine's compaction threshold is derived at
|
||||
* `context_window − COMPACTION_HEADROOM` (see effectiveMaxContextLength), so under normal
|
||||
* operation the context is summarized before the remaining window ever nears the
|
||||
* MIN_OUTPUT_TOKENS floor; the floor only binds when compaction is disabled or the window
|
||||
* is misconfigured, where a deterministic small cap beats a provider rejection.
|
||||
* `undefined` = no positive cap configured: the key stays off the wire and the provider's
|
||||
* own remaining-window default applies (the existing `-1` contract). Without a configured
|
||||
* `contextWindow` the clamp is off entirely (see effectiveMaxOutputTokens).
|
||||
*
|
||||
* A hard clamp — the derived cap dropping below half the configured one — is announced
|
||||
* once per distinct value on stderr with the numbers involved, so a shaved `max_tokens`
|
||||
* is diagnosable from the log instead of surfacing only as an opaque failed/short turn.
|
||||
*/
|
||||
private effectiveMaxTokens(newMessages: OmniMessage[]): number | undefined {
|
||||
const estimatedInput = this.lastRequestTotal + approximateMessagesTokens(newMessages);
|
||||
const derived = effectiveMaxOutputTokens(
|
||||
this.configuredMaxTokens,
|
||||
this.contextWindow,
|
||||
estimatedInput,
|
||||
);
|
||||
if (
|
||||
derived !== undefined &&
|
||||
this.configuredMaxTokens !== undefined &&
|
||||
derived < this.configuredMaxTokens / 2 &&
|
||||
derived !== this.lastWarnedCap
|
||||
) {
|
||||
this.lastWarnedCap = derived;
|
||||
process.stderr.write(
|
||||
`[penguin] output cap clamped hard: max_tokens ${this.configuredMaxTokens} -> ${derived} ` +
|
||||
`(context_window ${this.contextWindow}, estimated input ${estimatedInput})\n`,
|
||||
);
|
||||
}
|
||||
return derived;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1072,7 +1121,7 @@ export class GenerativeModel implements LLMInterface {
|
||||
* and the terminal state is then returned as `LLMOutcome`:
|
||||
* - **Normal completion**: `finish()` closes out and produces `token_usage` (usage is only
|
||||
* produced in this case) → `completed`;
|
||||
* - **Idle timeout / network drop** (retryable errors like network/429/5xx/quota):
|
||||
* - **Idle timeout / network drop** (transport-shaped errors like network/429/5xx):
|
||||
* `finishInterrupted("timeout")` closes out, produces no usage → `timeout` (carrying
|
||||
* the error detail as `message` when a concrete error was caught), reconnected by
|
||||
* `context_engine` within the same run;
|
||||
@@ -1158,9 +1207,11 @@ export class GenerativeModel implements LLMInterface {
|
||||
// parse error) / aborted (user) / failed (other). null means it ended normally.
|
||||
let outcome: LLMOutcome | null = null;
|
||||
try {
|
||||
const it = this.openStream(uniMessage, ac.signal, this.requestConfig(params.thinkingLevel))[
|
||||
Symbol.asyncIterator
|
||||
]();
|
||||
const it = this.openStream(
|
||||
uniMessage,
|
||||
ac.signal,
|
||||
this.requestConfig(params.thinkingLevel, params.newMessages),
|
||||
)[Symbol.asyncIterator]();
|
||||
for (;;) {
|
||||
// The interruption check must happen **before pulling from upstream**: the user may
|
||||
// interrupt while this generator is suspended at the `yield` below (the typical case —
|
||||
@@ -1224,10 +1275,9 @@ export class GenerativeModel implements LLMInterface {
|
||||
// branch as a belt — isRetryableError itself already refuses auth signals.
|
||||
outcome = { status: "auth", errorMessage: describeError(error) };
|
||||
} else if (isRetryableError(error)) {
|
||||
// Network drop / transient provider rejection -> needs reconnection. The detail
|
||||
// (e.g. "403 … (insufficient_user_quota)") rides on the outcome so observability
|
||||
// (request_end -> the Cost center's errors panel) shows the real reason behind a
|
||||
// retried request, not just "timeout".
|
||||
// Transport-shaped failure (network drop, 429/5xx) -> labeled timeout. The detail
|
||||
// rides on the outcome so observability (request_end -> the Cost center's errors
|
||||
// panel) shows the real reason behind a retried request, not just "timeout".
|
||||
outcome = { status: "timeout", errorMessage: describeError(error) };
|
||||
} else if ((error as { name?: string })?.name === "AbortError") {
|
||||
outcome = { status: "aborted" }; // Fallback: an unexpected abort (neither timeout nor user)
|
||||
@@ -1264,6 +1314,10 @@ export class GenerativeModel implements LLMInterface {
|
||||
// Normal completion: backfill stop + the complete model_msg, and produce token_usage.
|
||||
for (const msg of translator.finish()) yield msg;
|
||||
const requestTokens = translator.getRequestTokens();
|
||||
// The provider-measured context size, feeding the next request's output-cap clamp
|
||||
// (see effectiveMaxTokens). Only a completed request updates it: an interrupted or
|
||||
// failed attempt was never committed, so the context did not grow.
|
||||
this.lastRequestTotal = requestTokens.total;
|
||||
this.sessionTokens = addTokenCounts(this.sessionTokens, requestTokens);
|
||||
yield tokenUsage(this.sessionTokens, requestTokens);
|
||||
return { status: "completed" };
|
||||
@@ -1289,6 +1343,12 @@ export class GenerativeModel implements LLMInterface {
|
||||
this.toolCallIds.markUsed(p.tool_call_id);
|
||||
}
|
||||
}
|
||||
// The injected history is context this object's first request carries without any
|
||||
// token_usage having measured it: re-seed the input-size tracker (prefix + history
|
||||
// estimate) so the output-cap clamp (effectiveMaxTokens) doesn't reason from an empty
|
||||
// context on a resumed session. The first completed request replaces this with the
|
||||
// real total.
|
||||
this.lastRequestTotal = this.baseInputTokens + approximateMessagesTokens(history);
|
||||
this.client.setHistory(groupHistoryToUniMessages(history));
|
||||
}
|
||||
|
||||
|
||||
@@ -15,10 +15,20 @@ export {
|
||||
isMalformedJsonParseError,
|
||||
isIncompleteStreamError,
|
||||
isRetryableError,
|
||||
isQuotaExhaustedError,
|
||||
isAuthenticationError,
|
||||
mapThinkingLevel,
|
||||
toolDefinitionsToSchemas,
|
||||
buildUniConfig,
|
||||
} from "./generative-model.js";
|
||||
export { ToolCallIdAllocator, stripToolCallIdSuffix } from "./tool-call-ids.js";
|
||||
export {
|
||||
DEFAULT_CONTEXT_WINDOW,
|
||||
OUTPUT_SAFETY_MARGIN,
|
||||
MIN_OUTPUT_TOKENS,
|
||||
COMPACTION_HEADROOM,
|
||||
resolveContextWindow,
|
||||
approximateTokens,
|
||||
approximateMessagesTokens,
|
||||
effectiveMaxOutputTokens,
|
||||
effectiveMaxContextLength,
|
||||
} from "./context-limits.js";
|
||||
|
||||
@@ -24,7 +24,7 @@ import {
|
||||
saveProjectConfig,
|
||||
setVaultEntry,
|
||||
} from "../src/index.js";
|
||||
import { effectiveMaxContextLength, metaMaxTokens } from "../src/agent.js";
|
||||
import { metaMaxTokens } from "../src/agent.js";
|
||||
import { mapThinkingLevel } from "../src/llm/index.js";
|
||||
import { stubProviderKeys } from "./provider-keys.js";
|
||||
import type { EnvironmentConfig, EnvironmentServices, SubagentRunner } from "../src/interfaces.js";
|
||||
@@ -79,15 +79,8 @@ afterEach(async () => {
|
||||
await fs.rm(tmpRoot, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe("effectiveMaxContextLength (compaction threshold clamped to the model window)", () => {
|
||||
it("clamps to 75% of a small model window; leaves big/unknown windows and off untouched", () => {
|
||||
expect(effectiveMaxContextLength(128000, 32768)).toBe(24576); // small window: clamp to 75%
|
||||
expect(effectiveMaxContextLength(128000, 200000)).toBe(128000); // ample window: unchanged
|
||||
expect(effectiveMaxContextLength(-1, 32768)).toBe(-1); // off: no clamping
|
||||
expect(effectiveMaxContextLength(0, 32768)).toBe(0); // off: no clamping
|
||||
expect(effectiveMaxContextLength(128000, "unknown")).toBe(128000); // unknown window: no clamping
|
||||
});
|
||||
});
|
||||
// effectiveMaxContextLength moved to llm/context-limits.ts; its derivation (window-derived
|
||||
// compaction threshold, issue #218) is covered in test/context-limits.test.ts.
|
||||
|
||||
describe("metaMaxTokens (meta-request budget tightened by the per-model cap)", () => {
|
||||
it("keeps the budget unless the per-model cap is smaller; never raises it", () => {
|
||||
|
||||
@@ -0,0 +1,202 @@
|
||||
/**
|
||||
* Window-derived request limits (llm/context-limits.ts, issue #218): the per-request
|
||||
* output-cap arithmetic, the input-size estimator it feeds on, and the window-derived
|
||||
* compaction threshold. All pure functions — no timers, no network.
|
||||
*/
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
COMPACTION_HEADROOM,
|
||||
DEFAULT_CONTEXT_WINDOW,
|
||||
MIN_OUTPUT_TOKENS,
|
||||
MIN_USABLE_CONTEXT_WINDOW,
|
||||
OUTPUT_SAFETY_MARGIN,
|
||||
approximateMessagesTokens,
|
||||
approximateTokens,
|
||||
effectiveMaxContextLength,
|
||||
effectiveMaxOutputTokens,
|
||||
resolveContextWindow,
|
||||
} from "../src/llm/context-limits.js";
|
||||
import { toolCallOutput, userText } from "../src/omnimessage/index.js";
|
||||
import type { OmniMessage } from "../src/omnimessage/index.js";
|
||||
|
||||
describe("constant derivation (one tunable: OUTPUT_SAFETY_MARGIN)", () => {
|
||||
it("derives the floor and headroom from the margin so their invariants cannot drift", () => {
|
||||
// Floor below the margin: a floored request still fits the window when the input
|
||||
// estimate is accurate. Headroom above the margin: at the compaction trigger the
|
||||
// summary request keeps a usable output budget (~1k), not just the floor.
|
||||
expect(MIN_OUTPUT_TOKENS).toBe(Math.floor(OUTPUT_SAFETY_MARGIN / 2));
|
||||
expect(COMPACTION_HEADROOM).toBe(OUTPUT_SAFETY_MARGIN * 2);
|
||||
expect(COMPACTION_HEADROOM - OUTPUT_SAFETY_MARGIN).toBeGreaterThanOrEqual(1000);
|
||||
});
|
||||
});
|
||||
|
||||
describe("resolveContextWindow", () => {
|
||||
it("takes a plausible configured window at face value, anything else is unconfigured", () => {
|
||||
expect(resolveContextWindow(32768)).toBe(32768);
|
||||
expect(resolveContextWindow(MIN_USABLE_CONTEXT_WINDOW)).toBe(MIN_USABLE_CONTEXT_WINDOW);
|
||||
expect(resolveContextWindow(undefined)).toBeUndefined();
|
||||
expect(resolveContextWindow("unknown")).toBeUndefined();
|
||||
expect(resolveContextWindow(0)).toBeUndefined();
|
||||
expect(resolveContextWindow(-1)).toBeUndefined();
|
||||
expect(resolveContextWindow(Number.NaN)).toBeUndefined();
|
||||
// Below the sane minimum = a typo'd window (no real model sits under 4096): treated as
|
||||
// unconfigured rather than clamping every request against a bogus number.
|
||||
expect(resolveContextWindow(2048)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("approximateTokens (character heuristic, not a tokenizer)", () => {
|
||||
it("counts ASCII at ~4 chars/token and non-ASCII at 1 token/char", () => {
|
||||
expect(approximateTokens("")).toBe(0);
|
||||
expect(approximateTokens("abcd")).toBe(1);
|
||||
expect(approximateTokens("abcde")).toBe(2); // ceil(5/4)
|
||||
// CJK errs high on purpose: underestimating input risks a provider 400.
|
||||
expect(approximateTokens("你好世界")).toBe(4);
|
||||
expect(approximateTokens("ab你好")).toBe(3); // ceil(2/4) + 2
|
||||
});
|
||||
});
|
||||
|
||||
describe("approximateMessagesTokens", () => {
|
||||
it("estimates text-bearing payloads from their serialized size", () => {
|
||||
const text = userText("x".repeat(400));
|
||||
const est = approximateMessagesTokens([text]);
|
||||
// ~100 tokens of body + the JSON envelope: a loose sanity band, not an exact figure
|
||||
// (the heuristic's contract is "close and erring high").
|
||||
expect(est).toBeGreaterThanOrEqual(100);
|
||||
expect(est).toBeLessThan(150);
|
||||
// Tool outputs go through the same serialized-payload path; a genuinely huge text
|
||||
// output still counts as text.
|
||||
const big = toolCallOutput({
|
||||
output: "y".repeat(120_000),
|
||||
toolCallId: "c1",
|
||||
stopReason: "completed",
|
||||
});
|
||||
expect(approximateMessagesTokens([big])).toBeGreaterThanOrEqual(30_000);
|
||||
});
|
||||
|
||||
it("gives image payloads a flat allowance instead of counting data-URI characters", () => {
|
||||
const image = {
|
||||
type: "model_msg",
|
||||
payload: {
|
||||
type: "image_url",
|
||||
role: "user",
|
||||
image_url: `data:image/png;base64,${"A".repeat(100_000)}`,
|
||||
},
|
||||
} as unknown as OmniMessage;
|
||||
const est = approximateMessagesTokens([image]);
|
||||
// 100k base64 chars would be ~25k "tokens" by the char heuristic; the flat allowance
|
||||
// stays in the low thousands so one pasted screenshot cannot floor the output cap.
|
||||
expect(est).toBeLessThan(2000);
|
||||
expect(est).toBeGreaterThan(1000);
|
||||
});
|
||||
|
||||
it("counts a tool output's `images` data URLs at the flat allowance, not as base64 text", () => {
|
||||
// read_image-style outputs carry the image as tool_call_output.images (data URLs). A
|
||||
// 1 MB base64 string serialized as text would estimate ~262k "tokens" (~163x over)
|
||||
// and floor the NEXT request's cap even on a 128k window.
|
||||
const withImage = toolCallOutput({
|
||||
output: "Read image OK (1024x768).",
|
||||
toolCallId: "c1",
|
||||
stopReason: "completed",
|
||||
images: [`data:image/png;base64,${"A".repeat(1_000_000)}`],
|
||||
});
|
||||
const est = approximateMessagesTokens([withImage]);
|
||||
expect(est).toBeLessThan(2000); // flat image allowance + a small text payload
|
||||
expect(est).toBeGreaterThan(1600 - 1);
|
||||
// Two images: two allowances.
|
||||
const twoImages = toolCallOutput({
|
||||
output: "ok",
|
||||
toolCallId: "c2",
|
||||
stopReason: "completed",
|
||||
images: [
|
||||
`data:image/png;base64,${"A".repeat(500_000)}`,
|
||||
`data:image/jpeg;base64,${"B".repeat(500_000)}`,
|
||||
],
|
||||
});
|
||||
expect(approximateMessagesTokens([twoImages])).toBeLessThan(3400);
|
||||
expect(approximateMessagesTokens([twoImages])).toBeGreaterThan(3200 - 1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("effectiveMaxOutputTokens (per-request output clamp)", () => {
|
||||
it("is a no-op for big-window models: the configured cap comes back unchanged", () => {
|
||||
expect(effectiveMaxOutputTokens(32000, 1_000_000, 50_000)).toBe(32000);
|
||||
expect(effectiveMaxOutputTokens(32000, DEFAULT_CONTEXT_WINDOW, 10_000)).toBe(32000);
|
||||
});
|
||||
|
||||
it("keeps the no-explicit-cap contract: unset or non-positive stays undefined", () => {
|
||||
expect(effectiveMaxOutputTokens(undefined, 32768, 1000)).toBeUndefined();
|
||||
expect(effectiveMaxOutputTokens(-1, 32768, 1000)).toBeUndefined();
|
||||
expect(effectiveMaxOutputTokens(0, 32768, 1000)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not clamp without a configured window: no hard cap from an assumption", () => {
|
||||
// An entry without context_window used to derive a clamp from the assumed 128000;
|
||||
// with compaction disabled that pinned the cap to the floor past ~127k of real
|
||||
// context on a model that used to work. Unconfigured window = configured cap as-is.
|
||||
expect(effectiveMaxOutputTokens(32000, undefined, 500_000)).toBe(32000);
|
||||
expect(effectiveMaxOutputTokens(32000, undefined, 10)).toBe(32000);
|
||||
});
|
||||
|
||||
it("clamps so estimated input + cap + margin fits the window (the issue #218 report)", () => {
|
||||
// The reported failure: window 32768, configured cap 32000, prompt ~769 tokens —
|
||||
// the fixed cap overflowed the window on the very first request.
|
||||
const cap = effectiveMaxOutputTokens(32000, 32768, 769)!;
|
||||
expect(cap).toBe(32768 - 769 - OUTPUT_SAFETY_MARGIN);
|
||||
expect(769 + cap + OUTPUT_SAFETY_MARGIN).toBeLessThanOrEqual(32768);
|
||||
// The margin binds exactly: one token less input buys one token more cap.
|
||||
expect(effectiveMaxOutputTokens(32000, 32768, 768)).toBe(cap + 1);
|
||||
});
|
||||
|
||||
it("floors at MIN_OUTPUT_TOKENS instead of emitting a non-positive cap (degenerate case)", () => {
|
||||
// Remaining window smaller than the floor — compaction should have fired long before
|
||||
// this point; the deterministic floor beats sending max_tokens <= 0.
|
||||
expect(effectiveMaxOutputTokens(32000, 32768, 32_500)).toBe(MIN_OUTPUT_TOKENS);
|
||||
expect(effectiveMaxOutputTokens(32000, 32768, 99_999)).toBe(MIN_OUTPUT_TOKENS);
|
||||
// The floor stays below the safety margin, so a floored request still fits the window
|
||||
// whenever the input estimate is accurate.
|
||||
expect(MIN_OUTPUT_TOKENS).toBeLessThanOrEqual(OUTPUT_SAFETY_MARGIN);
|
||||
});
|
||||
|
||||
it("only ever lowers a cap: a configured cap below the floor is never raised", () => {
|
||||
// A pinned meta budget (metaMaxTokens can hand a cap as small as the user pinned it)
|
||||
// must come back verbatim — the clamp lowers caps, it never raises them.
|
||||
expect(effectiveMaxOutputTokens(128, 1_000_000, 100)).toBe(128);
|
||||
expect(effectiveMaxOutputTokens(128, 32768, 99_999)).toBe(128); // degenerate: still the configured cap
|
||||
});
|
||||
});
|
||||
|
||||
describe("effectiveMaxContextLength (window-derived compaction threshold)", () => {
|
||||
it("caps the configured threshold at context_window − COMPACTION_HEADROOM", () => {
|
||||
// 32k window: compaction now fires at ~30.7k instead of never (the 128000 default was
|
||||
// unreachable inside the window).
|
||||
expect(effectiveMaxContextLength(128000, 32768)).toBe(32768 - COMPACTION_HEADROOM);
|
||||
expect(effectiveMaxContextLength(128000, 200000)).toBe(128000); // ample window: unchanged
|
||||
expect(effectiveMaxContextLength(8000, 32768)).toBe(8000); // tighter user setting wins
|
||||
// A window at the sanity minimum still derives a positive threshold, so a usable
|
||||
// window can never flip the value into the "<=0 disables" contract.
|
||||
expect(effectiveMaxContextLength(128000, MIN_USABLE_CONTEXT_WINDOW)).toBe(
|
||||
MIN_USABLE_CONTEXT_WINDOW - COMPACTION_HEADROOM,
|
||||
);
|
||||
});
|
||||
|
||||
it("derives from the 128000 default when the window is unconfigured or implausible", () => {
|
||||
expect(effectiveMaxContextLength(128000, undefined)).toBe(
|
||||
DEFAULT_CONTEXT_WINDOW - COMPACTION_HEADROOM,
|
||||
);
|
||||
expect(effectiveMaxContextLength(128000, "unknown")).toBe(
|
||||
DEFAULT_CONTEXT_WINDOW - COMPACTION_HEADROOM,
|
||||
);
|
||||
// A typo'd tiny window counts as unconfigured (resolveContextWindow's sanity
|
||||
// threshold): the threshold derives from the assumption instead of collapsing to a
|
||||
// near-zero value that would compact after every request.
|
||||
expect(effectiveMaxContextLength(128000, 2048)).toBe(
|
||||
DEFAULT_CONTEXT_WINDOW - COMPACTION_HEADROOM,
|
||||
);
|
||||
});
|
||||
|
||||
it("keeps <=0 as 'compaction disabled'", () => {
|
||||
expect(effectiveMaxContextLength(-1, 32768)).toBe(-1); // off: no clamping
|
||||
expect(effectiveMaxContextLength(0, 32768)).toBe(0); // off: no clamping
|
||||
});
|
||||
});
|
||||
@@ -1910,14 +1910,15 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => {
|
||||
expect((abort!.payload as { reason?: string }).reason).toContain("llm request error");
|
||||
});
|
||||
|
||||
it("a quota-403 (classified timeout) retries within the default cap and succeeds", async () => {
|
||||
it("a bare 403 (classified failed) retries within the default cap and succeeds", async () => {
|
||||
let calls = 0;
|
||||
const llm: LLMInterface = {
|
||||
async *streamGenerate() {
|
||||
calls += 1;
|
||||
// Two quota rejections (GenerativeModel classifies them as timeout), then success —
|
||||
// attempt 3 is within the default cap of 5.
|
||||
if (calls <= 2) return { status: "timeout" };
|
||||
// Two provider 403 rejections (GenerativeModel labels them failed — there is no
|
||||
// quota/message heuristic anymore), then success: every non-auth failure rides
|
||||
// the ladder, and attempt 3 is within the default cap of 5.
|
||||
if (calls <= 2) return { status: "failed", errorMessage: "403 forbidden" };
|
||||
yield assistantText("recovered");
|
||||
yield tokenUsage(emptyTokenCounts(), {
|
||||
cache_read: 0,
|
||||
@@ -2020,15 +2021,16 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => {
|
||||
expect(engine.skipReconnectWait()).toBe(false);
|
||||
});
|
||||
|
||||
it("reconnectDelayMs: exponential-with-ceiling ladder (defaults: 250ms base, 30s cap)", () => {
|
||||
// The default cap (5) walks the first five steps — 250+500+1000+2000+4000 ≈ 7.75s of
|
||||
// total patience; the formula keeps climbing to the 30s ceiling for larger caps.
|
||||
const ladder = [1, 2, 3, 4, 5].map((n) => reconnectDelayMs(250, 30_000, n));
|
||||
expect(ladder).toEqual([250, 500, 1000, 2000, 4000]);
|
||||
expect(ladder.reduce((a, b) => a + b, 0)).toBe(7750);
|
||||
expect([6, 7, 8].map((n) => reconnectDelayMs(250, 30_000, n))).toEqual([8000, 16000, 30000]);
|
||||
it("reconnectDelayMs: exponential-with-ceiling ladder (defaults: 2s base, 30s cap)", () => {
|
||||
// The default cap (5) walks 2s/4s/8s/16s and hits the 30s ceiling on the fifth wait —
|
||||
// ≈ 60s of total patience (issue #218: the old 250ms base burned the whole ladder in
|
||||
// ~7.75s, faster than a provider restart or rate-limit window can recover). Every wait
|
||||
// sits at or above the hosts' 2s countdown floor, so each retry is visible.
|
||||
const ladder = [1, 2, 3, 4, 5].map((n) => reconnectDelayMs(2000, 30_000, n));
|
||||
expect(ladder).toEqual([2000, 4000, 8000, 16000, 30000]);
|
||||
expect(ladder.reduce((a, b) => a + b, 0)).toBe(60_000);
|
||||
// Past the ceiling the delay stays pinned (no overflow, no further growth).
|
||||
expect(reconnectDelayMs(250, 30_000, 12)).toBe(30_000);
|
||||
expect([6, 7].map((n) => reconnectDelayMs(2000, 30_000, n))).toEqual([30_000, 30_000]);
|
||||
// The cap also applies when the base itself exceeds it.
|
||||
expect(reconnectDelayMs(50_000, 30_000, 1)).toBe(30_000);
|
||||
});
|
||||
|
||||
+200
-94
@@ -21,12 +21,12 @@ import type { LLMOutcome, ThinkingLevelName } from "../src/interfaces.js";
|
||||
import {
|
||||
EventTranslator,
|
||||
GenerativeModel,
|
||||
MIN_OUTPUT_TOKENS,
|
||||
ToolCallIdAllocator,
|
||||
buildUniConfig,
|
||||
isAuthenticationError,
|
||||
isIncompleteStreamError,
|
||||
isMalformedJsonParseError,
|
||||
isQuotaExhaustedError,
|
||||
isRetryableError,
|
||||
mapThinkingLevel,
|
||||
mergeOmniToUniMessage,
|
||||
@@ -1077,29 +1077,37 @@ describe("config helpers", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("isRetryableError", () => {
|
||||
it("treats 429, 408 and 5xx as retryable", () => {
|
||||
describe("isRetryableError (the timeout-vs-failed label; the engine retries every non-auth error)", () => {
|
||||
it("labels 429, 408 and 5xx transport-shaped (timeout)", () => {
|
||||
expect(isRetryableError({ status: 429 })).toBe(true);
|
||||
expect(isRetryableError({ status: 408 })).toBe(true); // Request Timeout (transient)
|
||||
expect(isRetryableError({ status: 500 })).toBe(true);
|
||||
expect(isRetryableError({ statusCode: 503 })).toBe(true);
|
||||
});
|
||||
|
||||
it("treats 4xx auth/param errors as non-retryable", () => {
|
||||
it("labels provider 4xx rejections failed — including quota-coded and bare 403s", () => {
|
||||
// The label is taxonomy, not policy: `failed` is in the engine's RETRY_STATUSES, so
|
||||
// every one of these retries on the ladder and fails after exhaustion. There is no
|
||||
// quota detector anymore — a 402/403 with a quota code or subscription message
|
||||
// classifies exactly like a bare one.
|
||||
expect(isRetryableError({ status: 400 })).toBe(false);
|
||||
expect(isRetryableError({ status: 401 })).toBe(false);
|
||||
expect(isRetryableError({ status: 403 })).toBe(false);
|
||||
expect(isRetryableError({ status: 404 })).toBe(false);
|
||||
expect(isRetryableError({ status: 403, code: "insufficient_user_quota" })).toBe(false);
|
||||
expect(isRetryableError({ status: 402, code: "insufficient_quota" })).toBe(false);
|
||||
expect(isRetryableError({ status: 403, message: "no active subscription" })).toBe(false);
|
||||
// 401 is auth (terminal), not merely failed-shaped.
|
||||
expect(isRetryableError({ status: 401 })).toBe(false);
|
||||
});
|
||||
|
||||
it("treats network error codes as retryable", () => {
|
||||
it("labels network error codes transport-shaped", () => {
|
||||
expect(isRetryableError({ code: "ECONNRESET" })).toBe(true);
|
||||
expect(isRetryableError({ code: "ETIMEDOUT" })).toBe(true);
|
||||
expect(isRetryableError(new Error("socket hang up"))).toBe(true);
|
||||
expect(isRetryableError(new Error("request timeout"))).toBe(true);
|
||||
});
|
||||
|
||||
it("treats undici transport failures as retryable, probing the cause chain", () => {
|
||||
it("labels undici transport failures transport-shaped, probing the cause chain", () => {
|
||||
// Node fetch wraps a dropped connection as TypeError("terminated") with the real code on
|
||||
// `cause` — exactly the field observed: "terminated: other side closed (UND_ERR_SOCKET)".
|
||||
expect(
|
||||
@@ -1120,7 +1128,7 @@ describe("isRetryableError", () => {
|
||||
expect(isRetryableError(new TypeError("fetch failed"))).toBe(true);
|
||||
expect(isRetryableError(new TypeError("terminated: other side closed"))).toBe(true);
|
||||
expect(isRetryableError(new TypeError("terminated"))).toBe(false);
|
||||
// Provider verdict copy that merely contains the word must NOT retry.
|
||||
// Provider verdict copy that merely contains the word is failed-shaped, not transport.
|
||||
expect(isRetryableError(new Error("Request terminated by content filter"))).toBe(false);
|
||||
// Classic Node codes wrapped one level down are found too.
|
||||
expect(
|
||||
@@ -1132,42 +1140,16 @@ describe("isRetryableError", () => {
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("treats provider quota/subscription exhaustion as retryable (tight allowlist)", () => {
|
||||
// The field case: a gateway 403 with an OpenAI-compatible body code.
|
||||
expect(isRetryableError({ status: 403, code: "insufficient_user_quota" })).toBe(true);
|
||||
expect(isRetryableError({ status: 403, error: { code: "insufficient_user_quota" } })).toBe(
|
||||
true,
|
||||
);
|
||||
// Anthropic SDK shape: `error` holds the whole response body.
|
||||
expect(
|
||||
isRetryableError({ status: 403, error: { error: { code: "insufficient_quota" } } }),
|
||||
).toBe(true);
|
||||
expect(isRetryableError({ status: 402, code: "insufficient_quota" })).toBe(true);
|
||||
// Status-less quota code (some gateways surface only the provider code).
|
||||
expect(isRetryableError({ code: "insufficient_user_quota" })).toBe(true);
|
||||
// 403 with a quota/subscription message but no machine-readable code.
|
||||
expect(isRetryableError({ status: 403, message: "no active subscription" })).toBe(true);
|
||||
expect(isRetryableError({ status: 403, message: "订阅额度不足" })).toBe(true);
|
||||
// A plain 403/400/404 without quota signals stays non-retryable.
|
||||
expect(isRetryableError({ status: 403, message: "permission denied" })).toBe(false);
|
||||
expect(isRetryableError({ status: 400, code: "insufficient_user_quota" })).toBe(false);
|
||||
expect(isRetryableError({ status: 404 })).toBe(false);
|
||||
// 401 keeps its auth classification even with a quota-looking message.
|
||||
expect(isRetryableError({ status: 401, message: "quota" })).toBe(false);
|
||||
});
|
||||
|
||||
it("auth wins over the quota heuristic: a definitive credential signal is never retried", () => {
|
||||
// The adversarial case: a 403 whose BODY carries a definitive auth code but whose
|
||||
// MESSAGE mentions a subscription (SDKs routinely put the body's message on
|
||||
// err.message). The quota keyword fallback must not swallow it — auth is checked
|
||||
// first, so it classifies failed (dead credential) instead of burning every reconnect.
|
||||
it("auth always wins: a definitive credential signal is terminal, never merely a label", () => {
|
||||
// A 403 whose BODY carries a definitive auth code but whose MESSAGE mentions a
|
||||
// subscription (SDKs routinely put the body's message on err.message): the explicit
|
||||
// credential signal decides, keeping the dead credential off the retry ladder.
|
||||
const err = Object.assign(new Error("subscription key invalid"), {
|
||||
status: 403,
|
||||
error: { code: "invalid_api_key" },
|
||||
});
|
||||
expect(isAuthenticationError(err)).toBe(true);
|
||||
expect(isQuotaExhaustedError(err)).toBe(false); // belt: an auth signal is never quota
|
||||
expect(isRetryableError(err)).toBe(false); // never retried, never reclassified
|
||||
expect(isRetryableError(err)).toBe(false);
|
||||
// Same with the auth signal on the cause chain.
|
||||
expect(
|
||||
isRetryableError(
|
||||
@@ -1176,9 +1158,12 @@ describe("isRetryableError", () => {
|
||||
}),
|
||||
),
|
||||
).toBe(false);
|
||||
// A bare 403 carries no credential signal: NOT auth (it rides the retry ladder as
|
||||
// failed; see the engine tests).
|
||||
expect(isAuthenticationError({ status: 403, message: "forbidden" })).toBe(false);
|
||||
});
|
||||
|
||||
it("does not retry abort or unknown local errors", () => {
|
||||
it("labels abort and unknown local errors failed-shaped", () => {
|
||||
const abort = new Error("aborted");
|
||||
abort.name = "AbortError";
|
||||
expect(isRetryableError(abort)).toBe(false);
|
||||
@@ -1188,44 +1173,6 @@ describe("isRetryableError", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("isQuotaExhaustedError", () => {
|
||||
it("matches only 402/403 (or status-less) errors carrying a known quota code", () => {
|
||||
expect(isQuotaExhaustedError({ status: 403, code: "insufficient_user_quota" })).toBe(true);
|
||||
expect(isQuotaExhaustedError({ status: 402, error: { code: "insufficient_quota" } })).toBe(
|
||||
true,
|
||||
);
|
||||
expect(isQuotaExhaustedError({ code: "insufficient_quota" })).toBe(true);
|
||||
// The code may sit on the cause chain (wrapped by a higher layer).
|
||||
expect(
|
||||
isQuotaExhaustedError(
|
||||
new Error("request failed", {
|
||||
cause: { status: 403, code: "insufficient_user_quota" },
|
||||
}),
|
||||
),
|
||||
).toBe(true);
|
||||
expect(
|
||||
isQuotaExhaustedError(
|
||||
Object.assign(new Error("request failed"), {
|
||||
status: 403,
|
||||
cause: { code: "insufficient_user_quota" },
|
||||
}),
|
||||
),
|
||||
).toBe(true);
|
||||
// Any other status keeps its own classification.
|
||||
expect(isQuotaExhaustedError({ status: 429, code: "insufficient_quota" })).toBe(false);
|
||||
expect(isQuotaExhaustedError({ status: 401, code: "insufficient_quota" })).toBe(false);
|
||||
});
|
||||
|
||||
it("matches a 403 whose message names quota/subscription, but not a plain 403", () => {
|
||||
expect(isQuotaExhaustedError({ status: 403, message: "monthly quota exceeded" })).toBe(true);
|
||||
expect(isQuotaExhaustedError({ status: 403, message: "订阅额度不足" })).toBe(true);
|
||||
expect(isQuotaExhaustedError({ status: 403, message: "forbidden" })).toBe(false);
|
||||
// The message shortcut is 403-only: a status-less error needs the explicit code.
|
||||
expect(isQuotaExhaustedError({ message: "quota exceeded" })).toBe(false);
|
||||
expect(isQuotaExhaustedError(null)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isAuthenticationError", () => {
|
||||
it("classifies HTTP 401 as auth, own or wrapped", () => {
|
||||
expect(isAuthenticationError({ status: 401 })).toBe(true);
|
||||
@@ -1400,6 +1347,150 @@ describe("GenerativeModel per-request thinking level", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("GenerativeModel per-request output cap (window clamp, issue #218)", () => {
|
||||
// Same capturing pattern as the thinking-level suite: the openStream seam receives the
|
||||
// per-request resolved UniConfig, whose max_tokens is the value that would go on the
|
||||
// wire. The fake stream's usage_metadata drives lastRequestTotal between requests.
|
||||
function windowModel(opts: {
|
||||
maxTokens?: number;
|
||||
contextWindow?: number;
|
||||
promptTokens?: number;
|
||||
}): { model: GenerativeModel; configs: (UniConfig | undefined)[] } {
|
||||
const configs: (UniConfig | undefined)[] = [];
|
||||
class WindowModel extends GenerativeModel {
|
||||
protected override openStream(
|
||||
_uni: UniMessage,
|
||||
_signal: AbortSignal,
|
||||
config?: UniConfig,
|
||||
): AsyncIterable<UniEvent> {
|
||||
configs.push(config);
|
||||
return (async function* () {
|
||||
yield ev({
|
||||
content_items: [{ type: "text", text: "ok" }],
|
||||
finish_reason: "stop",
|
||||
usage_metadata: {
|
||||
cached_tokens: 0,
|
||||
prompt_tokens: opts.promptTokens ?? 1,
|
||||
thoughts_tokens: 0,
|
||||
response_tokens: 1,
|
||||
},
|
||||
});
|
||||
})();
|
||||
}
|
||||
}
|
||||
const model = new WindowModel({
|
||||
modelId: "claude-sonnet-4-6",
|
||||
tools: [],
|
||||
...(opts.maxTokens !== undefined ? { maxTokens: opts.maxTokens } : {}),
|
||||
...(opts.contextWindow !== undefined ? { contextWindow: opts.contextWindow } : {}),
|
||||
});
|
||||
return { model, configs };
|
||||
}
|
||||
|
||||
async function drainAll(gen: AsyncGenerator<OmniMessage, LLMOutcome | void>): Promise<void> {
|
||||
let res = await gen.next();
|
||||
while (!res.done) res = await gen.next();
|
||||
}
|
||||
|
||||
it("is a no-op for a big configured window: the configured cap goes out unchanged", async () => {
|
||||
const { model, configs } = windowModel({ maxTokens: 32000, contextWindow: 1_000_000 });
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("hi")] }));
|
||||
expect(configs[0]?.max_tokens).toBe(32000);
|
||||
});
|
||||
|
||||
it("never clamps without a configured window — even when the measured context is huge", async () => {
|
||||
// No contextWindow on the entry: a hard cap must not be derived from the 128000
|
||||
// assumption. A large-window model that omits context_window (and has compaction
|
||||
// disabled) would otherwise get its outputs floored past the assumed mark.
|
||||
const { model, configs } = windowModel({ maxTokens: 32000, promptTokens: 200_000 });
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("a")] }));
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("b")] }));
|
||||
expect(configs[0]?.max_tokens).toBe(32000);
|
||||
expect(configs[1]?.max_tokens).toBe(32000); // measured 200k context, still no clamp
|
||||
});
|
||||
|
||||
it("counts a tool output's base64 image at the flat allowance: the next request is not floored", async () => {
|
||||
// Regression for the review repro: a read_image-style tool output carrying a 1 MB
|
||||
// data-URL image used to be serialized into the estimate (~262k "tokens"), flooring
|
||||
// request 2's max_tokens to the minimum even on a 128k window.
|
||||
const { model, configs } = windowModel({
|
||||
maxTokens: 32000,
|
||||
contextWindow: 128_000,
|
||||
promptTokens: 1000,
|
||||
});
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("read the screenshot")] }));
|
||||
const imageOutput = toolCallOutput({
|
||||
output: "Read image OK (1024x768).",
|
||||
toolCallId: "c1",
|
||||
stopReason: "completed",
|
||||
images: [`data:image/png;base64,${"A".repeat(1_000_000)}`],
|
||||
});
|
||||
await drainAll(model.streamGenerate({ newMessages: [imageOutput] }));
|
||||
expect(configs[1]?.max_tokens).toBe(32000); // flat image allowance: nowhere near the window
|
||||
});
|
||||
|
||||
it("clamps the first request of a small-window model below the window (the vLLM 400)", async () => {
|
||||
// The issue #218 report: window 32768 with the seeded cap 32000 — the fixed cap
|
||||
// overflowed the window on the very first request. The clamped value must leave the
|
||||
// estimated input plus safety margin inside the window while staying positive.
|
||||
const { model, configs } = windowModel({ maxTokens: 32000, contextWindow: 32768 });
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("hello vllm")] }));
|
||||
const cap = configs[0]?.max_tokens ?? 0;
|
||||
expect(cap).toBeLessThan(32000);
|
||||
expect(cap).toBeGreaterThan(30000); // tiny prompt: only the estimate + margin is shaved off
|
||||
});
|
||||
|
||||
it("shrinks the cap as the measured context grows; floors instead of going non-positive", async () => {
|
||||
const { model, configs } = windowModel({
|
||||
maxTokens: 32000,
|
||||
contextWindow: 32768,
|
||||
promptTokens: 30_000,
|
||||
});
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("a")] }));
|
||||
// Request 2 reasons from request 1's real token_usage (total ≈ 30001): the remaining
|
||||
// window is ~1.7k, so the cap lands between the floor and 2k.
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("b")] }));
|
||||
const second = configs[1]?.max_tokens ?? 0;
|
||||
expect(second).toBeLessThan(2000);
|
||||
expect(second).toBeGreaterThanOrEqual(MIN_OUTPUT_TOKENS);
|
||||
expect(second).toBeLessThan(configs[0]?.max_tokens ?? 0);
|
||||
});
|
||||
|
||||
it("clamps to the deterministic floor once the window is exhausted (degenerate case)", async () => {
|
||||
// A measured context beyond the window (compaction disabled/misconfigured): the cap
|
||||
// pins at MIN_OUTPUT_TOKENS — never zero or negative, which providers reject outright.
|
||||
const { model, configs } = windowModel({
|
||||
maxTokens: 32000,
|
||||
contextWindow: 32768,
|
||||
promptTokens: 40_000,
|
||||
});
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("a")] }));
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("b")] }));
|
||||
expect(configs[1]?.max_tokens).toBe(MIN_OUTPUT_TOKENS);
|
||||
});
|
||||
|
||||
it("keeps the no-explicit-cap contract: without a configured cap nothing goes on the wire", async () => {
|
||||
// The provider's own default already bounds output by the remaining window; the clamp
|
||||
// must not invent a cap where the config said "none".
|
||||
const { model, configs } = windowModel({ contextWindow: 32768, promptTokens: 30_000 });
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("a")] }));
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("b")] }));
|
||||
expect(configs[0] !== undefined && "max_tokens" in configs[0]).toBe(false);
|
||||
expect(configs[1] !== undefined && "max_tokens" in configs[1]).toBe(false);
|
||||
});
|
||||
|
||||
it("setHistory seeds the input estimate, so a resumed session clamps its first request", async () => {
|
||||
const { model, configs } = windowModel({ maxTokens: 32000, contextWindow: 32768 });
|
||||
// ~30k estimated tokens of replayed history (120k ASCII chars): without the seed the
|
||||
// first request after resume would reason from an empty context and send ~31k.
|
||||
model.setHistory([userText("x".repeat(120_000))]);
|
||||
await drainAll(model.streamGenerate({ newMessages: [userText("continue")] }));
|
||||
const cap = configs[0]?.max_tokens ?? 0;
|
||||
expect(cap).toBeLessThan(2000);
|
||||
expect(cap).toBeGreaterThanOrEqual(MIN_OUTPUT_TOKENS);
|
||||
});
|
||||
});
|
||||
|
||||
describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () => {
|
||||
// Injects a controlled UniEvent stream through the protected openStream seam to verify the
|
||||
// outcome classification of timeout/network-drop/interrupt/error, without needing a real API.
|
||||
@@ -1637,7 +1728,9 @@ describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () =
|
||||
expect(outcome.errorMessage).toContain("invalid api key");
|
||||
expect(messages.map(typeOf)).not.toContain("token_usage");
|
||||
|
||||
// A genuinely non-retryable parameter error stays a plain failure.
|
||||
// A parameter error (400) labels failed — honestly reported, and still retried by the
|
||||
// engine like every non-auth failure (see engine.test.ts: it burns the ladder, then
|
||||
// surfaces verbatim).
|
||||
async function* paramError(): AsyncGenerator<UniEvent> {
|
||||
throw Object.assign(new Error("unknown parameter: max_output_tokens"), { status: 400 });
|
||||
}
|
||||
@@ -1648,10 +1741,10 @@ describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () =
|
||||
expect(outcome2.status).toBe("failed");
|
||||
});
|
||||
|
||||
it("an auth error dressed in quota language still funnels to status auth", async () => {
|
||||
// The adversarial case from the review: 403 + body code invalid_api_key + a message
|
||||
// mentioning the subscription. Must NOT classify as retryable quota (timeout): the
|
||||
// ordering fix keeps the dead credential out of the reconnect loop entirely.
|
||||
it("an explicit auth signal wins whatever the message says: status auth, never retried", async () => {
|
||||
// 403 + body code invalid_api_key + a message mentioning the subscription: the
|
||||
// explicit credential signal decides — auth is the one terminal status, and no
|
||||
// message vocabulary may pull a dead credential back onto the retry ladder.
|
||||
async function* dressedAuthError(): AsyncGenerator<UniEvent> {
|
||||
throw Object.assign(new Error("subscription key invalid"), {
|
||||
status: 403,
|
||||
@@ -1667,22 +1760,35 @@ describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () =
|
||||
expect(messages.map(typeOf)).not.toContain("token_usage");
|
||||
});
|
||||
|
||||
it("classifies provider quota exhaustion (403 + code) as timeout — the reconnect path", async () => {
|
||||
async function* quotaError(): AsyncGenerator<UniEvent> {
|
||||
throw Object.assign(new Error("订阅额度不足 no active subscription"), {
|
||||
it("labels a bare 403 as failed — a retryable status, no quota/message heuristics involved", async () => {
|
||||
// Every LLM error retries except auth: a 403 without a credential signal lands in
|
||||
// `failed` (which the engine reconnects on) instead of being probed for quota or
|
||||
// subscription vocabulary. The former quota detector is gone; a quota-coded provider
|
||||
// rejection classifies exactly the same way, and its real message still rides on the
|
||||
// outcome for observability.
|
||||
async function* bare403(): AsyncGenerator<UniEvent> {
|
||||
throw Object.assign(new Error("forbidden"), { status: 403 });
|
||||
}
|
||||
const model = new SeamModel(() => bare403());
|
||||
const { messages, outcome } = await drain(
|
||||
model.streamGenerate({ newMessages: [userText("go")] }),
|
||||
);
|
||||
expect(outcome.status).toBe("failed");
|
||||
expect(outcome.errorMessage).toContain("forbidden");
|
||||
expect(messages.map(typeOf)).not.toContain("token_usage");
|
||||
|
||||
async function* quotaCoded403(): AsyncGenerator<UniEvent> {
|
||||
throw Object.assign(new Error("no active subscription"), {
|
||||
status: 403,
|
||||
code: "insufficient_user_quota",
|
||||
});
|
||||
}
|
||||
const model = new SeamModel(() => quotaError());
|
||||
const { messages, outcome } = await drain(
|
||||
model.streamGenerate({ newMessages: [userText("go")] }),
|
||||
const model2 = new SeamModel(() => quotaCoded403());
|
||||
const { outcome: outcome2 } = await drain(
|
||||
model2.streamGenerate({ newMessages: [userText("go")] }),
|
||||
);
|
||||
expect(outcome.status).toBe("timeout");
|
||||
// The real reason rides on the outcome (request_end -> the Cost center's errors panel
|
||||
// shows it): a retried quota rejection must not surface as a bare "timeout".
|
||||
expect(outcome.errorMessage).toContain("insufficient_user_quota");
|
||||
expect(messages.map(typeOf)).not.toContain("token_usage");
|
||||
expect(outcome2.status).toBe("failed");
|
||||
expect(outcome2.errorMessage).toContain("insufficient_user_quota");
|
||||
});
|
||||
|
||||
it("classifies an undici transport drop (TypeError terminated, cause UND_ERR_SOCKET) as timeout", async () => {
|
||||
|
||||
@@ -103,7 +103,7 @@ Goal mode is the exception because its objective is re-injected as text every ro
|
||||
|
||||
## Automatic reconnect
|
||||
|
||||
Every LLM-side failure except `auth` triggers an in-run reconnect — `timeout` (network timeouts, transport disconnects, rate limits, 5xx, transient provider quota errors), `malformed` (truncated streams, JSON parse failures), and **`failed` as well**. `failed` retries even though the classifier judged it non-transient: that judgement is an allowlist of known codes, statuses and message vocabulary, so a gateway phrasing a transient fault its own way (`Upstream HTTP/2 stream failed`, say) lands there and used to kill the turn. Retrying a genuinely permanent error costs the ladder and ends the same way; aborting a transient one destroys the turn. Note this changes the *policy*, not the *taxonomy*: a `failed` request is still recorded as `failed` on its `request_end` and in the Cost center, rather than being relabelled a timeout. On a reconnect the engine re-sends the original input plus a `[turn_retried]` block carrying the previous partial output, so tools are never re-executed. Default limit is 5 reconnects with exponential backoff under a ceiling (base 250ms, cap 30s: 250ms, 500ms, 1s, 2s, 4s ≈ 7.75s of total patience — one shared schedule growing toward the slower retryable classes, with the first steps as fast as a transport blip needs); beyond that the turn settles as `failed`. Each failure's `request_end` announces the planned wait as `retry_in_ms` (same formula as the sleep) and stamps `attempt`, the authoritative 1-based ordinal of the request within its retry run (the CLI and Web App display it verbatim); the Web App renders the wait as a live countdown with "retry now" (skips the remaining wait via `Session.skipReconnectWait` — the attempt counter is unchanged) and "give up" (the ordinary abort; the engine's abort-during-backoff path ends the turn) controls; the CLI prints its own `[retry]` line. All three retryable statuses render identically — a retry the user cannot see is a stalled session with no explanation and no way out. A compaction request is an ordinary LLM request and by default retries on the same cap and ladder (an unusable summary draws on the same budget — see "Context compaction"); a compaction that gives up keeps the original context and tries again at the next trigger. Authentication errors are classified before any retry heuristic and never retry: the request ends with its own terminal status `auth` (only the model reference is fixed at Session creation — credentials are read from the current Project config when the Session loads), and the Web App disables that Session's composer until the model's credential is updated (which auto-unlocks it) or the notice is dismissed for a retry. Tool errors are never retried — they are fed back to the model as `tool_call_output` and the model decides what to do next.
|
||||
Every LLM-side failure except `auth` triggers an in-run reconnect — `timeout` (transport-shaped errors: network timeouts, transport disconnects, rate limits, 5xx), `malformed` (truncated streams, JSON parse failures), and **`failed` as well** — every provider rejection that isn't an explicit credential failure, bare 403s and quota/subscription errors included. The statuses are taxonomy, not policy: the classifier only picks the label, and a gateway phrasing a transient fault its own way (`Upstream HTTP/2 stream failed`, say) or a quota that refills mid-ladder retries exactly like a network drop. Retrying a genuinely permanent error costs the ladder and ends the same way; aborting a transient one destroys the turn. Note this changes the *policy*, not the *taxonomy*: a `failed` request is still recorded as `failed` on its `request_end` and in the Cost center, rather than being relabelled a timeout. On a reconnect the engine re-sends the original input plus a `[turn_retried]` block carrying the previous partial output, so tools are never re-executed. Default limit is 5 reconnects with exponential backoff under a ceiling (base 2s, cap 30s: 2s, 4s, 8s, 16s, 30s ≈ 60s of total patience — one shared schedule for every retryable class, sized so transient provider failures such as restarts and rate limits get a real recovery window instead of five retries burning out in about a second, and so every planned wait clears the Web App's 2s countdown floor and stays visible); beyond that the turn settles as `failed`. Each failure's `request_end` announces the planned wait as `retry_in_ms` (same formula as the sleep) and stamps `attempt`, the authoritative 1-based ordinal of the request within its retry run (the CLI and Web App display it verbatim); the Web App renders the wait as a live countdown with "retry now" (skips the remaining wait via `Session.skipReconnectWait` — the attempt counter is unchanged) and "give up" (the ordinary abort; the engine's abort-during-backoff path ends the turn) controls; the CLI prints its own `[retry]` line. All three retryable statuses render identically — a retry the user cannot see is a stalled session with no explanation and no way out. A compaction request is an ordinary LLM request and by default retries on the same cap and ladder (an unusable summary draws on the same budget — see "Context compaction"); a compaction that gives up keeps the original context and tries again at the next trigger. Authentication errors are classified before any retry heuristic and never retry: the request ends with its own terminal status `auth` (only the model reference is fixed at Session creation — credentials are read from the current Project config when the Session loads), and the Web App disables that Session's composer until the model's credential is updated (which auto-unlocks it) or the notice is dismissed for a retry. Tool errors are never retried — they are fed back to the model as `tool_call_output` and the model decides what to do next.
|
||||
|
||||
## Compaction
|
||||
|
||||
@@ -122,7 +122,7 @@ Three triggers (`compaction_begin.reason`):
|
||||
|
||||
| reason | Condition |
|
||||
| --- | --- |
|
||||
| `context` | last turn's `token_usage.request.total` ≥ `maxContextLength` (default 128000) |
|
||||
| `context` | last turn's `token_usage.request.total` ≥ `maxContextLength` (default 128000; the effective threshold is capped at the model's `context_window` − 2048, so a small-window model — a 32k local vLLM, say — compacts at ~30.7k instead of overflowing the window first; an entry without `context_window` derives from the assumed 128000 default) |
|
||||
| `turns` | Session turn count ≥ `maxSessionTurns` (default -1 = unlimited) |
|
||||
| `manual` | the user runs `/compact` or calls `session.compact()` |
|
||||
|
||||
|
||||
@@ -100,7 +100,7 @@ Task 运行期间,宿主可通过 `session.steer(input)` 排队一条用户消
|
||||
|
||||
## 自动重连
|
||||
|
||||
除 `auth` 外,LLM 侧的所有失败都会触发引擎内自动重连——`timeout`(网络超时、传输层断连、限流、5xx、瞬时的供应商额度错误)、`malformed`(流截断、JSON 解析失败),**以及 `failed`**。`failed` 也重试,尽管分类器判定它不是瞬时错误:那个判定本质是一张允许清单(已知错误码、状态码与消息措辞),所以用自己说法描述瞬时故障的网关(例如 `Upstream HTTP/2 stream failed`)会落到这一档,此前会直接终止本轮。重试一个真正的永久错误,代价是走完退避梯度后以同样的方式收场;而把瞬时错误直接中断,则毁掉这一轮。注意改的是**策略**而非**分类**:`failed` 请求在 `request_end` 与成本中心里仍然记为 `failed`,不会被改标成超时。重连时同一次 `run` 内重发原始输入,并附加 `[turn_retried]` 块携带上一次的部分输出,避免工具重复执行。默认最多重连 5 次,指数退避并设上限(基数 250ms、上限 30s:250ms、500ms、1s、2s、4s,总耐心约 7.75s——所有可重试类别共用一张时间表,向较慢的类别递增,头几步仍与传输层抖动所需的一样快);超限后该轮以 `failed` 收场。每次失败的 `request_end` 会以 `retry_in_ms` 宣告计划中的等待(与实际休眠同一公式)、以 `attempt` 标注这是本轮第几次尝试(权威序号,CLI 与 Web 的重试行直接显示它),Web App 据此实时倒计时,并提供「立即重试」(经 `Session.skipReconnectWait` 跳过剩余等待——重试计数不变)与「放弃」(普通中断;引擎的退避中中断路径结束本轮)两个内联按钮,CLI 则打印自己的 `[重试]` 行。三种可重试终态的渲染完全一致——用户看不见的重试,等于一次没有任何解释、也无从退出的卡顿。压缩请求是一次普通的 LLM 请求,默认沿用同一重连上限与退避阶梯(无效摘要也计入同一预算,见「上下文压缩」一节);压缩放弃后保留原上下文、等下一次触发再试。鉴权错误在任何重试启发式之前判定、从不重试:请求以专属终态 `auth` 收场(Session 锁定的只是模型引用,凭据在会话装载时取自当前 Project 配置),Web App 据此禁用该 Session 的输入框,直到该模型的凭据被更新(更新后自动解锁)或用户点击「重试」。工具错误从不重试——它们作为 `tool_call_output` 反馈给模型,由模型决定下一步。
|
||||
除 `auth` 外,LLM 侧的所有失败都会触发引擎内自动重连——`timeout`(传输形态的错误:网络超时、传输层断连、限流、5xx)、`malformed`(流截断、JSON 解析失败),**以及 `failed`**——凡不是明确凭据错误的供应商拒绝都在此列,普通 403 与配额/订阅错误也一样重试。终态只是分类而非策略:分类器只挑标签,用自己说法描述瞬时故障的网关(例如 `Upstream HTTP/2 stream failed`)或阶梯中途恢复的配额,都与断网一样走满同一阶梯。重试一个真正的永久错误,代价是走完退避梯度后以同样的方式收场;而把瞬时错误直接中断,则毁掉这一轮。注意改的是**策略**而非**分类**:`failed` 请求在 `request_end` 与成本中心里仍然记为 `failed`,不会被改标成超时。重连时同一次 `run` 内重发原始输入,并附加 `[turn_retried]` 块携带上一次的部分输出,避免工具重复执行。默认最多重连 5 次,指数退避并设上限(基数 2s、上限 30s:2s、4s、8s、16s、30s,总耐心约 60s——所有可重试类别共用一张时间表,按较慢的类别定基数:供应商重启、限流这类瞬时故障需要以秒计的恢复时间,旧的 250ms 基数约 7.75s 就烧完整个阶梯;每次计划等待也都达到 Web App 2s 的倒计时下限,重试始终可见);超限后该轮以 `failed` 收场。每次失败的 `request_end` 会以 `retry_in_ms` 宣告计划中的等待(与实际休眠同一公式)、以 `attempt` 标注这是本轮第几次尝试(权威序号,CLI 与 Web 的重试行直接显示它),Web App 据此实时倒计时,并提供「立即重试」(经 `Session.skipReconnectWait` 跳过剩余等待——重试计数不变)与「放弃」(普通中断;引擎的退避中中断路径结束本轮)两个内联按钮,CLI 则打印自己的 `[重试]` 行。三种可重试终态的渲染完全一致——用户看不见的重试,等于一次没有任何解释、也无从退出的卡顿。压缩请求是一次普通的 LLM 请求,默认沿用同一重连上限与退避阶梯(无效摘要也计入同一预算,见「上下文压缩」一节);压缩放弃后保留原上下文、等下一次触发再试。鉴权错误在任何重试启发式之前判定、从不重试:请求以专属终态 `auth` 收场(Session 锁定的只是模型引用,凭据在会话装载时取自当前 Project 配置),Web App 据此禁用该 Session 的输入框,直到该模型的凭据被更新(更新后自动解锁)或用户点击「重试」。工具错误从不重试——它们作为 `tool_call_output` 反馈给模型,由模型决定下一步。
|
||||
|
||||
## 上下文压缩(Compaction)
|
||||
|
||||
@@ -119,7 +119,7 @@ interface CompactionSettings {
|
||||
|
||||
| reason | 触发条件 |
|
||||
| --- | --- |
|
||||
| `context` | 上一轮 `token_usage.request.total` ≥ `maxContextLength`(默认 128000) |
|
||||
| `context` | 上一轮 `token_usage.request.total` ≥ `maxContextLength`(默认 128000;生效阈值不超过模型 `context_window` − 2048,小窗口模型——如 32k 的本地 vLLM——约在 30.7k 处压缩,而不是先撞上窗口硬限制;条目未配置 `context_window` 时按 128000 的假定默认值推导) |
|
||||
| `turns` | Session 轮数 ≥ `maxSessionTurns`(默认 -1,即不限) |
|
||||
| `manual` | 用户执行 `/compact` 或调用 `session.compact()` |
|
||||
|
||||
|
||||
@@ -99,10 +99,10 @@ Edit this file via the CLI (`penguin config model …`) or the Web Models page
|
||||
| `version` | `1` | Agent State version (a natural number), incremented on each successful optimization |
|
||||
| `system_prompt` | built-in template | Required; the only template with placeholder substitution |
|
||||
| `max_turns` | `-1` | Maximum LLM turns per Task (`-1` = unlimited; a positive integer caps the Task) |
|
||||
| `model.max_tokens` | `32000` | Output Token limit per Request (-1 = no cap, provider default) |
|
||||
| `model.max_tokens` | `32000` | Output Token ceiling per Request (-1 = no cap, provider default); each request clamps the effective value to the model's `context_window` minus the estimated input, so a small-window model never gets asked for more than fits |
|
||||
| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh`; the session default, overridable per-Task |
|
||||
| `model.timeoutMs` | `120000` | Per-Request timeout (milliseconds) |
|
||||
| `compaction.max_context_length` | `128000` | Context Token threshold that triggers compaction |
|
||||
| `compaction.max_context_length` | `128000` | Context Token threshold that triggers compaction; the effective threshold is capped at the model's `context_window` − 2048 so compaction fires before a small window overflows |
|
||||
| `compaction.max_session_turns` | `-1` | Cumulative Session turn threshold (`-1` = unlimited) |
|
||||
| `compaction.mode` | `summarize` | `summarize` / `discard` |
|
||||
| `compaction.prompt` | built-in template | Prompt used for summarize compaction |
|
||||
|
||||
@@ -99,10 +99,10 @@ output = 0.857143
|
||||
| `version` | `1` | Agent State 版本号(自然数),每次成功优化自增 |
|
||||
| `system_prompt` | 内置模板 | 必填;唯一进行占位符替换的模板 |
|
||||
| `max_turns` | `-1` | 单个 Task 的最大 LLM 轮数(`-1` 不限制,正整数为上限) |
|
||||
| `model.max_tokens` | `32000` | 单次输出 Token 上限(-1 不设上限,用服务商默认) |
|
||||
| `model.max_tokens` | `32000` | 单次输出 Token 天花板(-1 不设上限,用服务商默认);每次请求会把实际值收敛到模型 `context_window` 减估算输入以内,小窗口模型不会被索要放不下的输出 |
|
||||
| `model.thinking_level` | `medium` | `none` / `low` / `medium` / `high` / `xhigh`;作为会话默认档位,可被逐轮 Task 参数覆盖 |
|
||||
| `model.timeoutMs` | `120000` | 单次 Request 超时(毫秒) |
|
||||
| `compaction.max_context_length` | `128000` | 触发压缩的上下文 Token 阈值 |
|
||||
| `compaction.max_context_length` | `128000` | 触发压缩的上下文 Token 阈值;生效阈值不超过模型 `context_window` − 2048,压缩在小窗口溢出之前触发 |
|
||||
| `compaction.max_session_turns` | `-1` | Session 累计轮数阈值(`-1` 不限制) |
|
||||
| `compaction.mode` | `summarize` | `summarize` / `discard` |
|
||||
| `compaction.prompt` | 内置模板 | summarize 压缩使用的 Prompt |
|
||||
|
||||
@@ -60,7 +60,7 @@ interface LLMOutcome {
|
||||
| status | Meaning | Engine reaction |
|
||||
| --- | --- | --- |
|
||||
| `completed` | finished normally (token_usage already emitted) | proceed |
|
||||
| `timeout` | timeout / transport disconnect / transient provider quota error | auto-reconnect within the run |
|
||||
| `timeout` | timeout / transport disconnect | auto-reconnect within the run |
|
||||
| `malformed` | response parse failure | auto-reconnect within the run |
|
||||
| `failed` | an error the classifier did not judge transient (params, …) | auto-reconnect within the run as well — the status is still reported as `failed` |
|
||||
| `aborted` | user interrupt | stop, hand back to the user |
|
||||
|
||||
@@ -60,7 +60,7 @@ interface LLMOutcome {
|
||||
| status | 含义 | 引擎的反应 |
|
||||
| --- | --- | --- |
|
||||
| `completed` | 正常完成(已产出 token_usage) | 继续下一步 |
|
||||
| `timeout` | 超时/传输层断连/瞬时的供应商额度错误 | 同一 run 内自动重连 |
|
||||
| `timeout` | 超时/传输层断连 | 同一 run 内自动重连 |
|
||||
| `malformed` | 响应解析失败 | 同一 run 内自动重连 |
|
||||
| `failed` | 分类器未判定为瞬时的错误(参数等) | 同样在同一 run 内自动重连——状态本身仍如实上报为 `failed` |
|
||||
| `aborted` | 用户中断 | 停止交还用户 |
|
||||
|
||||
@@ -21,8 +21,8 @@ Each Project's available models are recorded in the hidden `.project_config.toml
|
||||
| --- | --- |
|
||||
| `provider` | Config group name; paired with `model_id` it forms the unique key |
|
||||
| `model_id` | Upstream request id |
|
||||
| `context_window` | Context window |
|
||||
| `max_tokens` | Optional per-model output cap (max output tokens per request). When set it overrides the Agent's `model.max_tokens`; unset inherits it. Lower it for small-context models: the per-Agent default (32000) cannot fit into e.g. a 32k window together with any prompt. Omitting the field on a Web full-table save clears it |
|
||||
| `context_window` | Context window (tokens). Load-bearing, not just display: each request's effective output cap and the compaction threshold are derived from it, so requests never ask for more output than the window still fits. Unset (or implausibly small, under 4096): the output clamp turns off and compaction derives from an assumed 128000 — set the real value for models with smaller windows |
|
||||
| `max_tokens` | Optional per-model output cap (max output tokens per request). When set it overrides the Agent's `model.max_tokens`; unset inherits it. The cap is a ceiling, not the literal wire value: each request sends `min(max_tokens, context_window − estimated input − safety margin)`, so small-window models work without hand-tuning it. Omitting the field on a Web full-table save clears it |
|
||||
| `client_type` | Protocol hint (e.g. `openai`); inferred by AgentHub from the model id when omitted |
|
||||
| `display_name` | Display name |
|
||||
| `vision` | Whether image input is supported, default true |
|
||||
@@ -77,6 +77,13 @@ The preset catalog also carries OpenRouter's free tier: `:free` model variants (
|
||||
|
||||
Some models in the preset catalog: deepseek-v4-pro / deepseek-v4-flash, gemini-3.1-pro-preview, claude-opus-4-8 / claude-sonnet-4-6, gpt-5.5, glm-5.2, kimi-k2.6, qwen3.8-max (not exhaustive).
|
||||
|
||||
## Local / self-hosted OpenAI-compatible endpoints (e.g. vLLM)
|
||||
|
||||
A local inference server is just a `custom` entry: `client_type = "openai"`, `base_url` pointing at the server (e.g. `http://127.0.0.1:8000/v1`), and the served model name as `model_id`. Two settings make it run smoothly:
|
||||
|
||||
- **Enable tool calling server-side.** For vLLM, start the server with `--enable-auto-tool-choice` and the `--tool-call-parser` matching your model (e.g. `hermes` for Qwen, `llama3_json` for Llama 3.x); without them tool calls arrive as plain text and the agent loop cannot execute anything.
|
||||
- **Set the entry's `context_window` to the server's real window** — for vLLM, the `--max-model-len` value (e.g. `32768`). The per-request output cap and the compaction threshold both derive from this window automatically: requests clamp `max_tokens` to what the window still fits, and compaction fires before the window overflows, so no hand-tuned `max_tokens` is needed. Left unset, the per-request output clamp is off and compaction assumes a 128000 window, so a smaller real window will reject requests.
|
||||
|
||||
## Thinking levels
|
||||
|
||||
Five levels: `none | low | medium | high | xhigh`, configured per Agent as `model.thinking_level` in `system_config.yaml`, default medium. The Web pickers offer `low` and above only (many models cannot disable thinking; `none` stays a valid stored value and still displays). The chat draft view offers a quick picker next to the model selector: a picked level is written back to the selected Agent's setting immediately (the switched-to level becomes that Agent's new default and applies from the next session). Inside an active session the thinking level is a **per-turn parameter**: the composer's picker lists only the levels and starts out showing the Agent config's level — while the user hasn't picked one it auto-follows the config (sends omit the level, so config edits keep taking effect); once picked, the level sticks for that session and rides on every subsequent send (it applies to that session's subsequent Tasks only and never writes back to the Agent config). See [Configuration](/configuration).
|
||||
|
||||
@@ -21,8 +21,8 @@ description: 经 AgentHub 单一网关接入模型,以 (provider, model_id)
|
||||
| --- | --- |
|
||||
| `provider` | 配置分组名,与 `model_id` 成对构成唯一键 |
|
||||
| `model_id` | 上游请求 id |
|
||||
| `context_window` | 上下文窗口 |
|
||||
| `max_tokens` | 可选的按模型输出上限(单次请求最大输出 Token 数)。设置后覆盖 Agent 的 `model.max_tokens`,缺省沿用;小上下文模型建议调低——按 Agent 的缺省值(32000)加上任意 Prompt 无法放进如 32k 的窗口。Web 整表保存时省略该字段即清除 |
|
||||
| `context_window` | 上下文窗口(Token 数)。不只用于展示:每次请求的实际输出上限与压缩阈值都由它推导,请求不会索要超出窗口剩余空间的输出。缺省(或小于 4096 的非常规值)时输出收敛关闭、压缩按 128000 假定推导——窗口更小的模型务必填真实值 |
|
||||
| `max_tokens` | 可选的按模型输出上限(单次请求最大输出 Token 数)。设置后覆盖 Agent 的 `model.max_tokens`,缺省沿用。该值是天花板而非逐字上线值:每次请求实际发送 `min(max_tokens, context_window − 估算输入 − 安全余量)`,小窗口模型无需手工调低。Web 整表保存时省略该字段即清除 |
|
||||
| `client_type` | 协议提示(如 `openai`);缺省由 AgentHub 按 model id 推断 |
|
||||
| `display_name` | 显示名 |
|
||||
| `vision` | 是否支持图像输入,默认 true |
|
||||
@@ -77,6 +77,13 @@ api_key = "sk-..."
|
||||
|
||||
预置目录中的部分模型:deepseek-v4-pro / deepseek-v4-flash、gemini-3.1-pro-preview、claude-opus-4-8 / claude-sonnet-4-6、gpt-5.5、glm-5.2、kimi-k2.6、qwen3.8-max 等(非完整清单)。
|
||||
|
||||
## 本地 / 自建 OpenAI 兼容端点(如 vLLM)
|
||||
|
||||
本地推理服务就是一条 `custom` 条目:`client_type = "openai"`、`base_url` 指向服务地址(如 `http://127.0.0.1:8000/v1`)、`model_id` 填服务端的模型名。两处设置决定运行是否顺畅:
|
||||
|
||||
- **服务端要开启工具调用。** vLLM 需以 `--enable-auto-tool-choice` 启动,并按模型选择对应的 `--tool-call-parser`(如 Qwen 用 `hermes`、Llama 3.x 用 `llama3_json`);不开启时工具调用会以纯文本返回,Agent 循环无法执行任何工具。
|
||||
- **条目的 `context_window` 填服务端的真实窗口**——vLLM 即 `--max-model-len` 的值(如 `32768`)。每次请求的输出上限与压缩阈值都会由该窗口自动推导:请求把 `max_tokens` 收敛到窗口剩余空间以内,压缩也会在撞上窗口硬限制之前触发,无需手工调低 `max_tokens`。不填时不做逐请求输出收敛、压缩按 128000 假定,真实窗口更小会导致请求被拒。
|
||||
|
||||
## 思考等级
|
||||
|
||||
思考等级共五档:`none | low | medium | high | xhigh`,按 Agent 在 `system_config.yaml` 的 `model.thinking_level` 配置,默认 medium。Web 拾取器只提供 `low` 及以上档位(多数模型不支持关闭思考;`none` 仍是合法的已存值,能正常显示)。对话草稿页在模型选择器旁提供快捷拾取器:选定档位立即写回所选 Agent 的该项配置(切换后的档位即成为该 Agent 的新默认,自下一个 Session 生效)。进行中的会话里,思考等级是**逐轮参数**:输入区拾取器只列出各档位,初始即显示 Agent 配置的档位——用户未手动选择时自动跟随配置下发(请求不携带档位,配置的修改持续生效);选定某档后即固定为该会话的档位,随之后每次发送携带(仅作用于该会话的后续 Task,不写回 Agent 配置)。见 [配置参考](/configuration)。
|
||||
|
||||
@@ -196,7 +196,7 @@ interface RequestEndPayload {
|
||||
error_message?: string; // error detail (LLMOutcome.errorMessage internally — one
|
||||
// name across the stack), non-completed only: the real
|
||||
// reason behind a retried/failed Request (e.g. a provider
|
||||
// quota code) — read by the Cost center's errors panel
|
||||
// error code) — read by the Cost center's errors panel
|
||||
attempt?: number; // 1-based ordinal of this request within its retry run (the
|
||||
// authoritative retry count): stamped on failures and on a
|
||||
// completion that needed retries; absent on a clean first try
|
||||
@@ -264,7 +264,7 @@ type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed" |
|
||||
| --- | --- | --- |
|
||||
| `completed` | finished normally | continue |
|
||||
| `aborted` | user interrupt | stop, hand back to the user |
|
||||
| `timeout` | LLM timeout / transport disconnect / transient provider quota error | LLM side only: auto-reconnect within the run |
|
||||
| `timeout` | LLM timeout / transport disconnect | LLM side only: auto-reconnect within the run |
|
||||
| `malformed` | parse failure / truncated stream | LLM side only: auto-reconnect within the run |
|
||||
| `failed` | an error the classifier did not judge transient (LLM); a tool error (Environment) | LLM side: auto-reconnect within the run as well — the status is still reported as `failed`. Environment side: the error is fed back to the model, never retried |
|
||||
| `auth` | the provider rejected the credentials | stop, hand back to the user — the one LLM status that never retries; hosts gate input until the model's API key is updated (credentials come from the current Project config) |
|
||||
|
||||
@@ -193,7 +193,7 @@ interface RequestEndPayload {
|
||||
// 不受影响;compaction_end 复用同一块——attempt 为最终尝试序号,error_message 为失败详情)
|
||||
error_message?: string; // 错误详情(内部 LLMOutcome.errorMessage,内外同名),仅非
|
||||
// completed 携带:被重试/失败的 Request 背后的真实原因
|
||||
// (如供应商额度码),供成本中心错误面板读取
|
||||
// (如供应商错误码),供成本中心错误面板读取
|
||||
attempt?: number; // 本次重试序列内的第几次请求(1 起,权威计数):失败请求
|
||||
// 与经重试后成功的请求携带;首发即成功不携带
|
||||
retry_in_ms?: number; // 计划中的重连等待(毫秒),仅当引擎将在本轮内重试时携带,
|
||||
@@ -260,7 +260,7 @@ type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed" |
|
||||
| --- | --- | --- |
|
||||
| `completed` | 正常完成 | 继续 |
|
||||
| `aborted` | 用户中断 | 停止并交还用户 |
|
||||
| `timeout` | LLM 超时/传输层断连/瞬时的供应商额度错误 | 仅 LLM 侧:同一 run 内自动重连 |
|
||||
| `timeout` | LLM 超时/传输层断连 | 仅 LLM 侧:同一 run 内自动重连 |
|
||||
| `malformed` | 响应解析失败/流截断 | 仅 LLM 侧:同一 run 内自动重连 |
|
||||
| `failed` | 分类器未判定为瞬时的错误(LLM 侧);工具执行出错(Environment 侧) | LLM 侧:同样在同一 run 内自动重连——该状态本身仍如实上报为 `failed`。Environment 侧:错误回灌给模型,从不重试 |
|
||||
| `auth` | 供应商拒绝了凭据 | 停止并交还用户——唯一从不重试的 LLM 终态;宿主据此禁用输入,直到该模型的 API key 被更新(凭据取自当前 Project 配置) |
|
||||
|
||||
Reference in New Issue
Block a user