From 5b22ad06dad40dbea435d9df95b7d924f40c4717 Mon Sep 17 00:00:00 2001 From: Yaowei Zheng Date: Wed, 29 Jul 2026 16:57:05 +0800 Subject: [PATCH] fix(core): retry on failed as well as timeout; only auth stops the run (#104) Co-authored-by: Claude Opus 5 (1M context) --- packages/cli/src/i18n.ts | 24 ++- packages/cli/src/render.ts | 11 +- packages/cli/test/render.test.ts | 24 +++ packages/core/src/engine/context-engine.ts | 137 ++++++++++++------ packages/core/src/interfaces.ts | 17 ++- packages/core/src/llm/generative-model.ts | 31 ++-- packages/core/src/state/default-config.ts | 2 +- packages/core/test/compaction.test.ts | 43 +++++- packages/core/test/engine.test.ts | 102 ++++++++++--- packages/core/test/llm.test.ts | 2 +- packages/docs/content/agent-loop.en.md | 6 +- packages/docs/content/agent-loop.zh.md | 6 +- packages/docs/content/architecture.en.md | 2 +- packages/docs/content/architecture.zh.md | 2 +- packages/docs/content/interfaces.en.md | 4 +- packages/docs/content/interfaces.zh.md | 4 +- packages/docs/content/message-flow.en.md | 2 +- packages/docs/content/message-flow.zh.md | 2 +- packages/docs/content/omni-message.en.md | 4 +- packages/docs/content/omni-message.zh.md | 4 +- packages/server/src/runtime/error-recorder.ts | 12 +- .../src/runtime/stream-error-watcher.ts | 83 +++++++++-- packages/server/src/services/trace-service.ts | 22 ++- packages/server/test/errors.test.ts | 58 +++++++- packages/server/test/session-manager.test.ts | 2 +- packages/server/test/trace-service.test.ts | 46 ++++++ packages/web/src/lib/omni/stream-model.ts | 35 +++-- packages/web/src/lib/strings-en.ts | 14 +- packages/web/src/lib/strings.ts | 15 +- packages/web/test/stream-model.test.ts | 61 ++++++++ 30 files changed, 603 insertions(+), 174 deletions(-) diff --git a/packages/cli/src/i18n.ts b/packages/cli/src/i18n.ts index 732dae8..9b6edcd 100644 --- a/packages/cli/src/i18n.ts +++ b/packages/cli/src/i18n.ts @@ -153,8 +153,12 @@ export interface Messages { }): string; /** Abort event label (may include a reason). */ abortLabel(reason?: string): string; - /** request_end ended with timeout/malformed: the engine retries (reconnect) carrying already-produced content; attempt is the retry count. */ - reconnectLabel(status: "timeout" | "malformed", attempt: number): string; + /** + * request_end ended with a status the engine reconnects on (`failed` / `timeout` / + * `malformed` — only `auth` is terminal): the engine retries carrying already-produced + * content; attempt is the retry count. + */ + reconnectLabel(status: "failed" | "timeout" | "malformed", attempt: number): string; /** compaction start event: indicates compaction in progress (mode is summarize/discard, reason is context/turns/manual). */ compactionStart(mode: string, reason: string): string; /** @@ -383,7 +387,13 @@ const en: Messages = { `[stats] context ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · ${s.elapsed} (${s.elapsedDelta})`, abortLabel: (reason) => `[abort]${reason ? `: ${reason}` : ""}`, reconnectLabel: (status, attempt) => - `[retry] ${status === "timeout" ? "connection timed out" : "response incomplete or unparseable"}; sending retry #${attempt}…`, + `[retry] ${ + status === "timeout" + ? "connection timed out" + : status === "malformed" + ? "response incomplete or unparseable" + : "the model provider returned an error" + }; sending retry #${attempt}…`, compactionStart: (mode, reason) => mode === "discard" ? `[compaction] discarding context (${reason})…` @@ -571,7 +581,13 @@ const zh: Messages = { `[统计信息] 上下文 ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · 用时 ${s.elapsed} (${s.elapsedDelta})`, abortLabel: (reason) => `[已中断]${reason ? `:${reason}` : ""}`, reconnectLabel: (status, attempt) => - `[重试] ${status === "timeout" ? "连接超时或网络中断" : "响应不完整或无法解析"},正在发起第 ${attempt} 次重试……`, + `[重试] ${ + status === "timeout" + ? "连接超时或网络中断" + : status === "malformed" + ? "响应不完整或无法解析" + : "模型服务返回错误" + },正在发起第 ${attempt} 次重试……`, compactionStart: (mode, reason) => mode === "discard" ? `[压缩] 正在丢弃旧上下文(${reason})……` diff --git a/packages/cli/src/render.ts b/packages/cli/src/render.ts index 9057778..352c12c 100644 --- a/packages/cli/src/render.ts +++ b/packages/cli/src/render.ts @@ -378,8 +378,8 @@ export class StreamRenderer { */ private taskFirstTsMs: number | null = null; private taskLastReqEndMs: number | null = null; - /** Terminal state (timeout/malformed) of the previous request: the next request_begin is a retry, at which point a notice is printed. */ - private pendingRetry: "timeout" | "malformed" | null = null; + /** Retryable terminal state (failed/timeout/malformed) of the previous request: the next request_begin is a retry, at which point a notice is printed. */ + private pendingRetry: "failed" | "timeout" | "malformed" | null = null; /** Number of retries already initiated (increments on consecutive failures, reset once a request completes normally). */ private reconnectRun = 0; @@ -693,7 +693,7 @@ export class StreamRenderer { this.out.write(`${formatAbort(payload as AbortPayload, this.t)}\n`); this.lastLineKey = null; } else if (payload.type === "request_begin") { - // The previous request ended in timeout/malformed -> this request is a retry + // The previous request ended in a retryable status -> this request is a retry // carrying [turn_retried]: printed when the retry **actually starts** (when // retries are exhausted, there's no retry after the last failure, only an abort // explaining why). @@ -720,7 +720,10 @@ export class StreamRenderer { this.hasUsage = true; } } - if (p.status === "timeout" || p.status === "malformed") { + // Every status the engine reconnects on, `failed` included — only `auth` is terminal. + // Leaving `failed` out would print nothing for a retry that is really happening, and + // reset the counter mid-ladder so a mixed run renumbers back to retry #1. + if (p.status === "failed" || p.status === "timeout" || p.status === "malformed") { this.pendingRetry = p.status; } else { this.pendingRetry = null; diff --git a/packages/cli/test/render.test.ts b/packages/cli/test/render.test.ts index a35dbd6..c78c8c7 100644 --- a/packages/cli/test/render.test.ts +++ b/packages/cli/test/render.test.ts @@ -303,6 +303,30 @@ describe("StreamRenderer", () => { expect(lines).not.toContain("retry #3"); }); + it("prints the retry line for a failed request too, and keeps counting across a mixed ladder", () => { + // The engine reconnects on `failed` as well, so the CLI has to say so — otherwise the + // session goes quiet for the whole ladder. And because `failed` used to reset the + // counter, a mixed timeout → failed → timeout run renumbered back to retry #1. + const { stream, text } = collector(); + const r = new StreamRenderer(stream, t); + r.handle(requestBegin()); + r.handle(requestEnd("failed", "Upstream HTTP/2 stream failed")); + r.handle(requestBegin()); // retry #1 begins + let lines = stripAnsi(text()); + expect(lines).toContain("the model provider returned an error"); + expect(lines).toContain("retry #1"); + r.handle(requestEnd("timeout")); + r.handle(requestBegin()); // retry #2 begins — the count does not restart + lines = stripAnsi(text()); + expect(lines).toContain("connection timed out"); + expect(lines).toContain("retry #2"); + // `auth` is terminal: the engine never retries it, so the next request_begin (a new run) + // must not be announced as a retry. + r.handle(requestEnd("auth", "401 invalid x-api-key")); + r.handle(requestBegin()); + expect(stripAnsi(text())).not.toContain("retry #3"); + }); + it("locks the screen to one streaming tool output; other messages queue until its stop", () => { const { stream, text } = collector(); const r = new StreamRenderer(stream, t); diff --git a/packages/core/src/engine/context-engine.ts b/packages/core/src/engine/context-engine.ts index 46a8f6a..b8f1046 100644 --- a/packages/core/src/engine/context-engine.ts +++ b/packages/core/src/engine/context-engine.ts @@ -183,11 +183,10 @@ export interface ContextEngineDeps { /** Ceiling (ms) for a single reconnect backoff wait. Defaults to 30000. */ reconnectBackoffMaxMs?: number; /** - * Maximum retries for a failing compaction request (instead of the shared - * `maxReconnects`). Compaction failure already has graceful semantics — the original - * context is kept and compaction retries on the next trigger — so it fails fast - * (~1.75s of backoff) rather than stalling the session through the full reconnect - * ladder. Defaults to 3. + * Maximum retries for a failing compaction request (instead of the shared `maxReconnects`; + * same statuses, shorter budget — see RETRY_STATUSES). Compaction failure is already + * graceful — the original context is kept and compaction retries at the next trigger — so a + * broken request fails fast (~1.75s of backoff). Defaults to 3. */ compactionMaxReconnects?: number; /** @@ -209,8 +208,8 @@ export type CompactAvailability = "ok" | "unsupported" | "empty" | "just_compact /** * Maximum summarize attempts when the compaction response is rejected as an invalid summary * (empty extracted text, or the model answered with tool calls — issue #83/#84). Deliberately - * separate from (and larger than) `compactionMaxReconnects`: that cap governs transport-level - * timeout/malformed attempts that were never committed and back off exponentially, while a + * separate from (and larger than) `compactionMaxReconnects`: that cap governs the retryable + * attempts that were never committed (failed/timeout/malformed) and back off exponentially, while a * rejection is a well-formed committed response — the request itself works, the model just * didn't produce a summary, so the repaired input is resent immediately with no backoff. And * since the compaction request keeps the session's toolset (the prefix cache must stay valid, @@ -317,6 +316,19 @@ function downgradeCarriedGoalInput(msg: OmniMessage): OmniMessage { return { ...msg, payload: { ...msg.payload, text: downgraded } as OmniMessage["payload"] }; } +/** + * LLM outcomes that reconnect in-run — everything but `auth`, which cannot be retried into + * working. `failed` is included on purpose: the classifier producing it is an allowlist of + * known transport codes and message vocabulary, so a gateway phrasing a transient fault its + * own way lands here; retrying a genuinely permanent error costs the ladder and ends the same + * way, while aborting a transient one destroys the turn. + * + * Both loops retry the same set. What differs is the budget: compaction runs on + * `compactionMaxReconnects` (shorter than the turn loop's), because a compaction that gives + * up keeps the original context and tries again at the next trigger. + */ +const RETRY_STATUSES: readonly StopReason[] = ["failed", "timeout", "malformed"]; + /** * Delay before reconnect attempt N (1-based): exponential growth from `base` with a hard * ceiling `max` — `min(base × 2^(N−1), max)`. With the defaults (250ms base, 30s ceiling, @@ -507,11 +519,18 @@ export class ContextEngine { } turnCount += 1; - // This turn's input. A timeout/malformed attempt is never committed to history by - // AgentHub (an abnormally interrupted stream doesn't land in history), so reconnect - // resends this turn's input unchanged, appending a `[turn_retried]` block carrying what - // the failed attempt already produced — the model continues from there instead of - // re-running tools; the tag is distinct from the user-interruption `[turn_aborted]`. + // This turn's input. The safety invariant behind resending it: **no retryable attempt + // is ever committed to AgentHub's history**. AgentHub appends a turn to `_history` only + // after its stream has been consumed to the end and validated, so every abnormal exit — + // `timeout`, `malformed` and `failed` alike, whether the stream was cut, the payload + // failed to parse, or the request was rejected outright — leaves history untouched. + // Nothing can therefore be duplicated or left as an unanswered tool_use by a reconnect; + // that is what makes it safe to resend this turn's input unchanged, appending a + // `[turn_retried]` block carrying what the failed attempt already produced — the model + // continues from there instead of re-running tools; the tag is distinct from the + // user-interruption `[turn_aborted]`. (The one attempt that IS committed — a fully + // delivered response whose finish_reason arrived — is forced to `completed` in + // GenerativeModel precisely so it never reaches this loop.) const failedTurns: TurnResult[] = []; let attemptInput = nextInput; let reconnects = 0; @@ -532,12 +551,10 @@ export class ContextEngine { yield* this.emitAbort("aborted by user"); return; } - // Non-retryable error: stop and hand control back to the user; the failure reason - // is written to the abort event / Trace. `auth` (credentials rejected) behaves - // exactly like `failed` here — direct stop, never the retry loop below (the - // classifier already refuses to mark auth retryable; this branch is the second - // belt) — hosts read the status from this turn's request_end to gate input. - if (turn.outcome.status === "failed" || turn.outcome.status === "auth") { + // `auth` is the one status that stops the run: a rejected credential cannot be retried + // into working, and hosts read it off the turn's request_end to gate input. Everything + // else — `failed` included — goes to the reconnect loop below. + if (turn.outcome.status === "auth") { this.pendingCarryOver = this.buildCarryOver(attemptInput, turn); yield* this.emitAbort(`llm request error: ${turn.outcome.message ?? "unknown"}`); return; @@ -545,17 +562,37 @@ export class ContextEngine { // Completed normally. if (turn.outcome.status === "completed") break; - // Only timeout / malformed remain: reconnect automatically within the same run. When - // retries are exhausted or the backoff is interrupted, the retry input is held as-is as - // carry-over (the original input is already written to Trace, so it isn't rewritten). - // The frontend surfaces the retry process and count via request_end(timeout|malformed) - // followed by the next request_begin. + // failed / timeout / malformed remain: reconnect automatically within the same run. + // + // `failed` retries too, even though the classifier judged it non-transient. That + // judgement is a hint, not a verdict: it is an allowlist of known network codes, + // statuses and message vocabulary, so every gateway that phrases a transient failure + // its own way falls through it — `Upstream HTTP/2 stream failed + // (upstream_http2_stream_error)` is plainly a transport fault and matched nothing. + // Retrying a genuinely permanent error costs the ladder and then ends the same way; + // aborting a transient one destroys the turn. So the *classification* stays honest + // (`failed` is still recorded as a real failure, not relabelled a timeout) while the + // *policy* retries it. Only `auth` is terminal, above. + // + // When retries are exhausted or the backoff is interrupted, the retry input is held + // as-is as carry-over (the original input is already written to Trace, so it isn't + // rewritten). The frontend surfaces the retry process and count via + // request_end(failed|timeout|malformed) followed by the next request_begin. failedTurns.push(turn); attemptInput = this.withRetriedTurns(nextInput, failedTurns); if (reconnects >= this.maxReconnects) { this.pendingCarryOver = attemptInput; - const reason = turn.outcome.status === "malformed" ? "malformed response" : "reconnect"; - yield* this.emitAbort(`${reason} failed after ${this.maxReconnects} retries`); + // This reason is user-visible (the Web App's error panel, the CLI's abort line) and + // is what observability persists as the error message, so it has to read as a + // sentence: what gave out, then how many attempts it took, then — for the one class + // that carries provider words worth showing — the detail, last. + const reason = + turn.outcome.status === "malformed" + ? `malformed response failed after ${this.maxReconnects} retries` + : turn.outcome.status === "failed" + ? `llm request failed after ${this.maxReconnects} retries: ${turn.outcome.message ?? "unknown"}` + : `reconnect failed after ${this.maxReconnects} retries`; + yield* this.emitAbort(reason); return; } reconnects += 1; @@ -740,14 +777,16 @@ export class ContextEngine { * follows instead). Announced on the failure's `request_end` as `retry_in_ms` so the * frontend can render a live countdown; shares `reconnectDelayMs` with `backoff`, so the * announced wait and the actual sleep cannot drift. The caps differ by loop: the turn - * loop passes `maxReconnects`, the compaction loop `compactionMaxReconnects`. + * loop passes `maxReconnects`, the compaction loop `compactionMaxReconnects`; `retries` must + * match what the calling loop actually does, or the announced countdown is a lie. */ private plannedRetryDelayMs( - status: StopReason, + outcome: LLMOutcome, reconnectsSoFar: number, cap: number, + retries: readonly StopReason[], ): number | undefined { - if (status !== "timeout" && status !== "malformed") return undefined; + if (!retries.includes(outcome.status)) return undefined; if (reconnectsSoFar >= cap) return undefined; return reconnectDelayMs( this.reconnectBackoffMs, @@ -868,7 +907,12 @@ export class ContextEngine { const stopEvt = requestEnd( outcome.status, outcome.message, - this.plannedRetryDelayMs(outcome.status, reconnectsSoFar, this.maxReconnects), + this.plannedRetryDelayMs( + outcome, + reconnectsSoFar, + this.maxReconnects, + RETRY_STATUSES, + ), ); queue.push(stopEvt); await this.write(stopEvt); @@ -896,8 +940,9 @@ export class ContextEngine { // added to this turn's ledger, and gets no paired output backfilled: such a tool_call // was never committed to history by AgentHub, so there's nothing to pair. This turn // must then end with a non-completed outcome (only interruption closure produces such - // a tool_call): timeout/malformed is cleaned up by reconnect resending the flatten - // carry-over, failed/auth/aborted exits directly. + // a tool_call): a retryable outcome (failed/timeout/malformed) is cleaned up by + // reconnect resending the flatten carry-over, while the run-ending ones + // (aborted/auth) exit directly. if (isCompleteModelMessage(msg) && msg.payload.type === "tool_call") { const tc = msg as OmniMessage; if (tc.payload.stop_reason !== "completed") continue; @@ -1128,9 +1173,9 @@ export class ContextEngine { * text and no tool calls in the response. An invalid summary is rejected: any tool calls the * model issued are answered with synthesized failed outputs (pairing repair, see the loop * body) and the repaired input is resent immediately, up to MAX_SUMMARY_REJECTIONS attempts, - * then the compaction fails. timeout/malformed reconnect via the existing retry mechanism - * under the compaction-specific cap (`compactionMaxReconnects`, tighter than the turn loop's - * ladder), collapsing to failed once retries are exhausted; on failure/abort, the original + * then the compaction fails. failed/timeout/malformed reconnect under the compaction-specific + * cap (`compactionMaxReconnects` — a shorter budget than the turn loop's, not a narrower + * set), collapsing to failed once retries are exhausted; on failure/abort, the original * context and Trace index are kept — it does not fall back to discard. The first **committed** * attempt absorbs `pendingToolOutputs` into the old context's history (issue #85): later * resends carry only the repairs and the Prompt, and the result's `committed` flag tells @@ -1167,7 +1212,7 @@ export class ContextEngine { // by a committed request: prepended to the retry input, and stashed as carry-over should // the compaction be abandoned first (see stashRepairs). let pendingRepairs: OmniMessage[] = []; - // Two independent retry budgets: transport-level timeout/malformed attempts (never + // Two independent retry budgets: retryable attempts (failed/timeout/malformed, never // committed) follow the compaction-specific reconnect cap with the exponential backoff // ladder; invalid-summary rejections (committed, well-formed responses that just aren't // summaries) get the larger dedicated cap and resend immediately — see @@ -1251,7 +1296,7 @@ export class ContextEngine { yield* this.emitCompactionEnd(reason, "summarize", "aborted"); return { status: "aborted", committed }; } - if (attempt.status === "failed" || attempt.status === "auth") { + if (attempt.status === "auth") { // `auth` folds into `failed` here: the compaction event pair keeps its // completed/failed/aborted set, the original context is kept, and the host learns // about the credential problem from the request's own terminal status (a turn-loop @@ -1260,11 +1305,11 @@ export class ContextEngine { yield* this.emitCompactionEnd(reason, "summarize", "failed"); return { status: "failed", committed }; } - // timeout / malformed: retried via reconnect — transport-level, never committed by + // failed / timeout / malformed: retried via reconnect — none of them is committed by // AgentHub (case B), so the input (any pending repairs included) is resent unchanged. - // Compaction uses its own, tighter cap (not the shared maxReconnects): a failed - // compaction keeps the original context and retries on the next trigger, so failing - // fast beats holding the session through the full exponential ladder. + // Compaction uses its own, tighter cap (not the shared maxReconnects): a compaction + // that ends up failing keeps the original context and tries again on the next trigger, + // so a short ladder here beats holding the session through the full one. if (reconnects >= this.compactionMaxReconnects) { this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "failed"); @@ -1345,9 +1390,10 @@ export class ContextEngine { res.value.status, res.value.message, this.plannedRetryDelayMs( - res.value.status, + res.value, reconnectsSoFar, this.compactionMaxReconnects, + RETRY_STATUSES, ), ), ); @@ -1423,10 +1469,11 @@ export class ContextEngine { /** * Builds the interruption resend content (carry-over, interruption cleanup) - * based on the LLM's terminal state. Used only for the **exit** cleanup of - * aborted / failed / auth (reconnect retry doesn't go through here — retry input is assembled by - * withRetriedTurns, appending `[turn_retried]` with the failed attempt's output, distinct - * from the user-interruption `[turn_aborted]`): + * based on the LLM's terminal state. Used only for the **exit** cleanup of the statuses that + * end the run: `aborted` and `auth` (a retried `failed` does not reach here, and neither + * does reconnect retry: retry input is assembled by withRetriedTurns, appending + * `[turn_retried]` with the failed attempt's output, distinct from the user-interruption + * `[turn_aborted]`): * - Model output completed (case A, outcome=completed): AgentHub already committed an * assistant turn containing `tool_call`, so it can only be resent as a structured * `tool_call_output` to pair with it (cannot flatten, or the already-committed tool_call diff --git a/packages/core/src/interfaces.ts b/packages/core/src/interfaces.ts index 83f99c3..a69ea78 100644 --- a/packages/core/src/interfaces.ts +++ b/packages/core/src/interfaces.ts @@ -147,13 +147,16 @@ export interface GenerativeModelParameters { * - `malformed`: AgentHub response failed JSON parsing, needs reconnect — also retried by * `context_engine`; * - `aborted`: user-initiated interruption — stop and hand back to the user; - * - `failed`: other non-retryable errors (params, etc.) — stop and hand back to the user - * (`message` provides the display text); - * - `auth`: the provider rejected the credentials (see `isAuthenticationError`) — behaves - * like `failed` in the engine (direct stop, never retried), but hosts key on the status - * to disable input until the model's API key is updated (only the model reference is - * fixed at Session creation; credentials come from the current Project config, so a key - * update lets the Session continue). + * - `failed`: an error the retry classifier did not judge transient (params, etc.) — still + * retried by `context_engine` within the same run (`message` provides the display text). + * The classification stays honest — this is reported as `failed`, not relabelled a + * timeout — while the *policy* retries it, because that classifier is an allowlist and a + * gateway phrasing a transient fault its own way lands here; + * - `auth`: the provider rejected the credentials (see `isAuthenticationError`) — the one + * status that stops the run outright, since no retry can turn a rejected credential into + * a working one; hosts also key on it to disable input until the model's API key is + * updated (only the model reference is fixed at Session creation; credentials come from + * the current Project config, so a key update lets the Session continue). * Docs: /docs/interfaces § "LLMOutcome semantics". */ export interface LLMOutcome { diff --git a/packages/core/src/llm/generative-model.ts b/packages/core/src/llm/generative-model.ts index 79ab7ae..415a5fa 100644 --- a/packages/core/src/llm/generative-model.ts +++ b/packages/core/src/llm/generative-model.ts @@ -12,13 +12,16 @@ * (observability/Token). * 3. Interruption/error handling: `finishInterrupted` first closes any open * streaming segments and backfills the complete message, then the output ends — never - * leaking a malformed structure. This interface **never retries internally** — retryable - * errors (network/transport drops, timeouts, 429/5xx, provider quota exhaustion, see + * leaking a malformed structure. This interface **never retries internally** — it only + * classifies, and `context_engine` owns the retry policy: errors recognised as retryable + * (network/transport drops, timeouts, 429/5xx, provider quota exhaustion, see * `isRetryableError`) end with `timeout`; AgentHub JSON parse errors end with `malformed`; - * both are handed to `context_engine` to reconnect within the same run. User interruption - * ends with `aborted`; non-retryable errors (parameters etc.) end with `failed`; - * credentials failures end with their own terminal status `auth` (see - * `isAuthenticationError`) — same engine behavior as `failed`, but hosts key on it. + * everything else the classifier does not recognise ends with `failed` — which the engine + * reconnects on all the same, because this classifier is an allowlist and a gateway + * wording a transient fault its own way falls through it. User interruption ends with + * `aborted`; credentials failures end with their own terminal status `auth` (see + * `isAuthenticationError`) — the one status the engine refuses to retry, and the one + * hosts key on to gate input. * * `context_engine` only consumes OmniMessage; all Uni* protocol details are encapsulated here. * Docs: /docs/interfaces § "The built-in implementation: GenerativeModel". @@ -1078,11 +1081,12 @@ export class GenerativeModel implements LLMInterface { * - **User interruption**: `finishInterrupted("aborted")` closes out, produces no usage → * `aborted`; * - **Credentials failure**: `finishInterrupted("auth")` closes out, produces no usage → - * `auth` (carrying `message`) — the engine stops exactly like `failed`, hosts key on - * the status to gate input until the model's API key is updated; - * - **Other non-retryable errors** (parameters etc.): `finishInterrupted("failed")` - * closes out, produces no usage → `failed` (carrying `message`), handed to - * `context_engine` to stop and return control to the user. + * `auth` (carrying `message`) — the one status the engine stops the run on, and the one + * hosts key on to gate input until the model's API key is updated; + * - **Every other error** (parameters etc., and input that never assembled into a + * request): `finishInterrupted("failed")` closes out, produces no usage → `failed` + * (carrying `message`), which `context_engine` reconnects on as well — the + * classification stays honest, the retry decision is the engine's. * * Timeout detection: the idle timer resets on every event received; once idle exceeds * `requestTimeoutMs`, the underlying stream is aborted and handled as needing reconnection @@ -1103,10 +1107,7 @@ export class GenerativeModel implements LLMInterface { try { uniMessage = mergeOmniToUniMessage(params.newMessages); } catch (err) { - return { - status: "failed", - message: describeError(err), - }; + return { status: "failed", message: describeError(err) }; } const translator = new EventTranslator(this.toolCallIds); diff --git a/packages/core/src/state/default-config.ts b/packages/core/src/state/default-config.ts index 4b11f04..0bb3ad8 100644 --- a/packages/core/src/state/default-config.ts +++ b/packages/core/src/state/default-config.ts @@ -112,7 +112,7 @@ Communicate with the user precisely and concisely, yet with warmth. Do not repea # System markers Some messages contain system-synthesized blocks written as \`[tag]...[/tag]\`, not user text to answer directly: - \`[turn_aborted]\`: the previous round was interrupted. Inside are the original request, your partial thinking/text, and the tool calls already issued with their results. Continue from where it left off; do not re-run tools whose results are already included. -- \`[turn_retried]\`: the previous attempt of this round failed on a transient error (timeout, disconnect, malformed response, or a temporary provider error) — the user did NOT interrupt — and this request is the automatic retry. Inside are your partial thinking/text and the tool calls already executed with their results. Continue from them; do not re-run tools whose results are already included. +- \`[turn_retried]\`: the previous attempt of this round did not get through (a timeout, a disconnect, a malformed response, or an error the provider returned) — the user did NOT interrupt — and this request is the automatic retry. Inside are your partial thinking/text and the tool calls already executed with their results. Continue from them; do not re-run tools whose results are already included. - \`[context_summary]\`: earlier conversation was compacted. This summary replaced the raw transcript and is its only record; treat it as established context and continue the task from it. - \`[user_steering]\`: a user message sent while you were still working, delivered between turns alongside tool results. It is not a new task: incorporate it immediately and adjust course within the current task. diff --git a/packages/core/test/compaction.test.ts b/packages/core/test/compaction.test.ts index 13096df..e65b8e7 100644 --- a/packages/core/test/compaction.test.ts +++ b/packages/core/test/compaction.test.ts @@ -12,8 +12,9 @@ * stay byte-identical, #84), and a completed response only counts as success with a valid * summary — non-empty extracted text and no tool calls (issue #83). A rejected response has its * tool calls answered by synthesized failed outputs (pairing repair) and is retried under a - * dedicated cap of 5 rejections; then the compaction fails. Transport timeout/malformed - * attempts keep the shared reconnect cap. + * dedicated cap of 5 rejections; then the compaction fails. Retryable attempts + * (failed/timeout/malformed) take the compaction-specific reconnect cap — the same set the + * turn loop retries, on a shorter budget; only `auth` stops at once. * - discard: deferred until task end if mid-task; sends no compaction request, just swaps in a new LLM instance directly. * - Process visibility: the compaction request's streamed output is never surfaced to the human, * only the paired compaction events are emitted; the dialogue is written to the old trace, and @@ -306,8 +307,8 @@ describe("context compaction", () => { { messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)], }, - // Compaction request fails (not retryable). - { messages: [], outcome: { status: "failed", message: "auth error" } }, + // Compaction request fails on the one status no ladder can fix (a rejected credential). + { messages: [], outcome: { status: "auth", message: "auth error" } }, // Original context is kept: the task continues, tool outputs feed back into the old instance as usual (context usage keeps growing). { messages: [assistantText("finished on old context"), usage(190, 340)] }, // Second trigger (context still over the limit) -> retries compaction at the boundary, this time succeeding. @@ -420,6 +421,40 @@ describe("context compaction", () => { expect(payloadTypes(llm1.calls[2]!)).toEqual(["text"]); }); + it("a failed compaction request takes the ladder and can recover on a later attempt", async () => { + // The compaction loop retries the same statuses the turn loop does: `failed` is where a + // transient fault lands whenever the classifier doesn't recognize the gateway's wording, + // and giving up here keeps the full context, so the next request re-triggers compaction + // against the same wall with less headroom. + const llm1 = new ScriptedLLM( + [ + { messages: [assistantText("answer"), usage(150, 150)] }, + { messages: [], outcome: { status: "failed", message: "502 upstream" } }, + { messages: [assistantText("[summary]recovered[/summary]")] }, + ], + "llm1", + ); + const llm2 = new ScriptedLLM([], "llm2"); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => llm2, + compactionMaxReconnects: 2, + reconnectBackoffMs: 1, + }); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + expect(compactionEvents(out)[1]).toMatchObject({ + type: "compaction_end", + status: "completed", + }); + // 1 turn request + 2 compaction attempts (the failed one, then the retry that succeeds). + expect(llm1.calls).toHaveLength(3); + // The retry resends the same input (tool results + prompt; only the prompt here). + expect(payloadTypes(llm1.calls[2]!)).toEqual(["text"]); + }); + it("the compaction loop uses its own cap, not the shared maxReconnects ladder", async () => { // Compaction failure has graceful semantics (the original context is kept, compaction // retries on the next trigger), so it fails fast: with compactionMaxReconnects=2, the diff --git a/packages/core/test/engine.test.ts b/packages/core/test/engine.test.ts index 31b5042..a244f4d 100644 --- a/packages/core/test/engine.test.ts +++ b/packages/core/test/engine.test.ts @@ -521,7 +521,7 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => { expect(texts.join("\n")).not.toContain("[turn_aborted]"); }); - it("downgrades a goal round's protocol in the [turn_aborted] transcript (LLM failure path)", async () => { + it("downgrades a goal round's protocol in the [turn_aborted] transcript (auth exit path)", async () => { // An aborted/failed goal round's input rides into the next task via flatten carry-over; // its [goal] protocol ("the system sends the next round automatically", the file rules) // is stale the moment the goal ends and must not re-enter the model as live instructions. @@ -541,7 +541,8 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => { if (++calls === 1) { yield partialText("start", ""); yield partialText("delta", "half a thought"); - return { status: "failed", message: "boom" }; + // `auth`: the one LLM status that still exits straight to the flatten path. + return { status: "auth", message: "boom" }; } yield assistantText("ok"); yield tokenUsage(emptyTokenCounts(), { @@ -1644,14 +1645,14 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { expect(nextRunTexts.join("\n")).not.toContain("[turn_aborted]"); }); - it("surfaces a non-retryable LLM failure (outcome=failed) as a graceful abort (run does not throw)", async () => { + it("a failed outcome retries like a timeout, then converges to a graceful abort (run does not throw)", async () => { let calls = 0; const inputs: OmniMessage[][] = []; const llm: LLMInterface = { - // The LLM must never throw an exception at the engine: a non-retryable error resolves - // by returning a failed outcome after closing the structure. A genuinely non-retryable - // failure nowadays is a parameter error (quota 403s retry as timeout, 401s carry - // code "auth" — both covered by their own tests below). + // The LLM must never throw an exception at the engine: an error resolves by returning + // a failed outcome after closing the structure. `failed` is still the honest + // classification for a parameter error — it is simply retried anyway, because the + // classifier cannot reliably tell a permanent 4xx from a gateway's transient one. // eslint-disable-next-line require-yield async *streamGenerate(params) { calls += 1; @@ -1663,31 +1664,81 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { workspaceDir: workspace, toolConfig: execCommandToolConfig(), }); - const engine = new ContextEngine({ llm, environment, reconnectBackoffMs: 0 }); + const engine = new ContextEngine({ + llm, + environment, + maxReconnects: 2, + reconnectBackoffMs: 0, + }); - // Must not throw; should gracefully converge to an abort. + // Must not throw; should gracefully converge to an abort once the ladder is spent. const all = await collectRun(engine, [userText("go")], allowAll); - expect(calls).toBe(1); // failed -> no retry. + expect(calls).toBe(3); // initial attempt + maxReconnects(2): `failed` takes the ladder now. const abort = all.find((m) => (m.payload as { type?: string }).type === "abort"); expect(abort).toBeDefined(); + // Asserted whole, not by fragments: this string is shown verbatim in the error panel and + // the CLI and is persisted as the error message, so its grammar is part of the contract. const reason = (abort!.payload as { reason?: string }).reason ?? ""; - expect(reason).toContain("llm request error"); - expect(reason).toContain("unknown parameter"); + expect(reason).toBe( + "llm request failed after 2 retries: 400 unknown parameter: max_output_tokens", + ); - // The failed turn's input is flattened and stashed; the next run resends it merged with - // the new input. + // The spent turn's input is stashed as carry-over; the next run (attempt index 3, after + // this run's three) resends it merged with the new input. await collectRun(engine, [userText("next")], allowAll); - const text = inputs[1]!.map((m) => (m.payload as { text?: string }).text ?? "").join("\n"); + const text = inputs[3]!.map((m) => (m.payload as { text?: string }).text ?? "").join("\n"); expect(text).toContain("go"); expect(text).toContain("next"); }); - it("an auth outcome stops immediately like failed: no retry, request_end carries status auth", async () => { + it("a failed request that succeeds on retry never reaches the user as an error", async () => { + // The point of retrying `failed`: the classifier is an allowlist, so a transient gateway + // fault phrased its own way ("Upstream HTTP/2 stream failed") lands here. It used to kill + // the turn; now the turn simply completes. + let calls = 0; + const llm: LLMInterface = { + async *streamGenerate() { + calls += 1; + if (calls === 1) { + return { + status: "failed" as const, + message: "Upstream HTTP/2 stream failed (upstream_http2_stream_error)", + }; + } + yield assistantText("recovered"); + return { status: "completed" as const }; + }, + }; + const engine = new ContextEngine({ + llm, + environment: new Environment({ + workspaceDir: workspace, + toolConfig: execCommandToolConfig(), + }), + reconnectBackoffMs: 0, + }); + const all = await collectRun(engine, [userText("go")], allowAll); + expect(calls).toBe(2); + expect(all.find((m) => (m.payload as { type?: string }).type === "abort")).toBeUndefined(); + // The failure is still classified `failed` on the wire — the retry is a policy decision, + // not a relabelling, so observability still sees a real failure rather than a "timeout". + const ends = all.filter((m) => (m.payload as { type?: string }).type === "request_end"); + expect(ends.map((m) => (m.payload as { status?: string }).status)).toEqual([ + "failed", + "completed", + ]); + // ...and it announces its retry wait like any other retryable failure, so the frontend + // countdown works for it too. + expect((ends[0]!.payload as { retry_in_ms?: number }).retry_in_ms).toBeGreaterThanOrEqual(0); + }); + + it("auth is the only LLM status that stops the run: no retry, request_end carries status auth", async () => { let calls = 0; const llm: LLMInterface = { // GenerativeModel classifies a 401/invalid_api_key as status "auth" (see - // llm.test.ts); the engine must stop directly — the auth/failed branch is the second - // belt keeping a dead credential out of the retry loop (the classifier is the first). + // llm.test.ts); the engine must stop directly — the auth branch is the second belt + // keeping a dead credential out of the retry loop (the classifier is the first). + // Every other failure, `failed` included, takes the ladder instead. // eslint-disable-next-line require-yield async *streamGenerate() { calls += 1; @@ -1701,7 +1752,7 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { const engine = new ContextEngine({ llm, environment, reconnectBackoffMs: 1 }); const all = await collectRun(engine, [userText("go")], allowAll); - expect(calls).toBe(1); // Auth behaves like failed: never enters the reconnect loop. + expect(calls).toBe(1); // A rejected credential cannot be retried into working. // The request's own terminal status is the host signal (streams to the web). const end = all.find((m) => (m.payload as { type?: string }).type === "request_end"); expect((end!.payload as { status?: string }).status).toBe("auth"); @@ -1970,7 +2021,7 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { expect(all.map((m) => (m.payload as { type?: string }).type)).not.toContain("abort"); }); - it("flatten carry-over (failed exit) includes the model's partial thinking and text (PRN-014)", async () => { + it("flatten carry-over (auth exit) includes the model's partial thinking and text (PRN-014)", async () => { let calls = 0; const inputs: OmniMessage[][] = []; const llm: LLMInterface = { @@ -1978,11 +2029,14 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { calls += 1; inputs.push(params.newMessages); if (calls === 1) { - // Before the non-retryable error, partial thinking and text were already produced - // (the LLM finishes them as complete messages, stop_reason failed). + // Before the terminal error, partial thinking and text were already produced (the + // LLM finishes them as complete messages). `auth` is the trigger because it is the + // one LLM status that still exits straight to the flatten path — `failed` now takes + // the reconnect ladder, whose carry-over is [turn_retried] instead (covered by the + // exhausted-retries test below). yield thinkingMessage("half-thought", "failed"); yield assistantText("half-text", "failed"); - return { status: "failed", message: "boom" }; + return { status: "auth", message: "boom" }; } yield assistantText("ok"); yield tokenUsage(emptyTokenCounts(), { @@ -2001,7 +2055,7 @@ describe("ContextEngine LLM timeout / network interruption (PRN-012)", () => { const engine = new ContextEngine({ llm, environment, reconnectBackoffMs: 0 }); await collectRun(engine, [userText("go")], allowAll); - expect(calls).toBe(1); // failed -> no retry, exits immediately. + expect(calls).toBe(1); // auth -> no retry, exits immediately. // Next run: the flattened carry-over contains the original input plus partial thinking/text // (both completed and incomplete messages are carried over). diff --git a/packages/core/test/llm.test.ts b/packages/core/test/llm.test.ts index c383e1f..609a485 100644 --- a/packages/core/test/llm.test.ts +++ b/packages/core/test/llm.test.ts @@ -1482,7 +1482,7 @@ describe("GenerativeModel.streamGenerate outcome classification (PRN-013)", () = return { messages, outcome: res.value as LLMOutcome }; } - it("returns failed (never throws) on a build failure such as empty input", async () => { + it("returns failed on a build failure such as empty input (never throws)", async () => { const model = new SeamModel((sig) => hang(sig)); const { messages, outcome } = await drain(model.streamGenerate({ newMessages: [] })); expect(outcome.status).toBe("failed"); // A mergeOmniToUniMessage failure converges to failed, never throws diff --git a/packages/docs/content/agent-loop.en.md b/packages/docs/content/agent-loop.en.md index 36ddfb7..1d3ff34 100644 --- a/packages/docs/content/agent-loop.en.md +++ b/packages/docs/content/agent-loop.en.md @@ -25,7 +25,7 @@ session.run(newMessages, { approve, signal }) │ │ Environment.executeTool ──► runs concurrently, │ │ │ output streams back │ │ └─ LLMOutcome: │ -│ timeout / malformed ──► reconnect within the turn │ +│ failed/timeout/malformed ──► reconnect within the turn │ │ (≤5, with [turn_retried]; tools not rerun) │ │ token_usage + request_end (at LLM-stream end; not waiting │ │ for tools) │ @@ -91,7 +91,7 @@ While a Task is running, the host can queue a user message with `session.steer(t ## Automatic reconnect -Only LLM-side `timeout` (network timeouts, transport disconnects such as a dropped socket, rate limits, 5xx, and transient provider quota/subscription errors like `insufficient_user_quota`) and `malformed` (truncated streams, JSON parse failures) trigger an in-run reconnect: the engine re-sends the original input plus a `[turn_retried]` block carrying the previous partial output, so tools are never re-executed. Default limit is 5 reconnects with exponential backoff under a ceiling (base 250ms, cap 30s: 250ms, 500ms, 1s, 2s, 4s ≈ 7.75s of total patience — one shared schedule growing toward the slower retryable classes, with the first steps as fast as a transport blip needs); beyond that the turn settles as `failed`. Each failure's `request_end` announces the planned wait as `retry_in_ms` (same formula as the sleep), which the Web App renders as a live countdown with "retry now" (skips the remaining wait via `Session.skipReconnectWait` — the attempt counter is unchanged) and "give up" (the ordinary abort; the engine's abort-during-backoff path ends the turn) controls. Compaction requests use their own tighter cap (3 retries): a failed compaction keeps the original context and retries on the next trigger, so failing fast beats stalling the session. Authentication errors are classified before any retry heuristic and never retry: the request ends with its own terminal status `auth` (only the model reference is fixed at Session creation — credentials are read from the current Project config when the Session loads), and the Web App disables that Session's composer until the model's credential is updated (which auto-unlocks it) or the notice is dismissed for a retry. Tool errors are never retried — they are fed back to the model as `tool_call_output` and the model decides what to do next. +Every LLM-side failure except `auth` triggers an in-run reconnect — `timeout` (network timeouts, transport disconnects, rate limits, 5xx, transient provider quota errors), `malformed` (truncated streams, JSON parse failures), and **`failed` as well**. `failed` retries even though the classifier judged it non-transient: that judgement is an allowlist of known codes, statuses and message vocabulary, so a gateway phrasing a transient fault its own way (`Upstream HTTP/2 stream failed`, say) lands there and used to kill the turn. Retrying a genuinely permanent error costs the ladder and ends the same way; aborting a transient one destroys the turn. Note this changes the *policy*, not the *taxonomy*: a `failed` request is still recorded as `failed` on its `request_end` and in the Cost center, rather than being relabelled a timeout. On a reconnect the engine re-sends the original input plus a `[turn_retried]` block carrying the previous partial output, so tools are never re-executed. Default limit is 5 reconnects with exponential backoff under a ceiling (base 250ms, cap 30s: 250ms, 500ms, 1s, 2s, 4s ≈ 7.75s of total patience — one shared schedule growing toward the slower retryable classes, with the first steps as fast as a transport blip needs); beyond that the turn settles as `failed`. Each failure's `request_end` announces the planned wait as `retry_in_ms` (same formula as the sleep), which the Web App renders as a live countdown with "retry now" (skips the remaining wait via `Session.skipReconnectWait` — the attempt counter is unchanged) and "give up" (the ordinary abort; the engine's abort-during-backoff path ends the turn) controls; the CLI prints its own `[retry]` line. All three retryable statuses render identically — a retry the user cannot see is a stalled session with no explanation and no way out. Compaction requests retry the same statuses under their own tighter cap (3 retries): a compaction that gives up keeps the original context and tries again at the next trigger, so a short ladder there beats stalling the session behind the full one. Authentication errors are classified before any retry heuristic and never retry: the request ends with its own terminal status `auth` (only the model reference is fixed at Session creation — credentials are read from the current Project config when the Session loads), and the Web App disables that Session's composer until the model's credential is updated (which auto-unlocks it) or the notice is dismissed for a retry. Tool errors are never retried — they are fed back to the model as `tool_call_output` and the model decides what to do next. ## Compaction @@ -116,7 +116,7 @@ Three triggers (`compaction_begin.reason`): Two modes: `summarize` (default) appends the compaction Prompt to the old context, extracts the `[summary]`, wraps it as a `[context_summary]` user text and continues in a **fresh model context**; `discard` simply drops the old context. System markers are written as `[tag]…[/tag]`; the earlier angle-bracket form (``, ``, …) is still recognized when reading old Traces and old persisted compaction prompts. Compaction rotates the [Trace file](/sessions-and-traces) (`_002`, `_003`, …) — one Trace file always equals one complete model context. `compactability()` probes feasibility before `session.compact()` (`ok | unsupported | empty | just_compacted`). -The compaction request keeps the session's toolset **unchanged** — the request prefix (tool list included) stays byte-identical to ordinary turns, so the provider's prompt cache remains valid at the moment the context is largest. Compaction still succeeds only with a valid summary: a response that calls a tool or whose extracted summary is empty is rejected — any tool calls are answered with synthesized failed outputs (keeping `tool_use`/`tool_result` pairing intact) and the repaired request is resent immediately (a rejection is model behavior, not a transport failure, so no backoff applies), up to 5 rejected attempts; then the compaction ends `failed`, keeping the original context and Trace file until the next trigger. Transport `timeout`/`malformed` attempts follow the compaction-specific reconnect cap and backoff ladder described under "Automatic reconnect" above. The first **committed** attempt — adopted or rejected — also absorbs whatever turn input was folded into the compaction request (mid-task tool results, or the carry-over a manual `/compact` folds in) into the old context's history: retries resend only the repairs and the Prompt, and the continuation after an abandoned compaction never resends the absorbed input. +The compaction request keeps the session's toolset **unchanged** — the request prefix (tool list included) stays byte-identical to ordinary turns, so the provider's prompt cache remains valid at the moment the context is largest. Compaction still succeeds only with a valid summary: a response that calls a tool or whose extracted summary is empty is rejected — any tool calls are answered with synthesized failed outputs (keeping `tool_use`/`tool_result` pairing intact) and the repaired request is resent immediately (a rejection is model behavior, not a transport failure, so no backoff applies), up to 5 rejected attempts; then the compaction ends `failed`, keeping the original context and Trace file until the next trigger. Retryable attempts (`failed`/`timeout`/`malformed`, the same set the turn loop retries) follow the compaction-specific reconnect cap and backoff ladder described under "Automatic reconnect" above; only `auth` stops the compaction at once. The first **committed** attempt — adopted or rejected — also absorbs whatever turn input was folded into the compaction request (mid-task tool results, or the carry-over a manual `/compact` folds in) into the old context's history: retries resend only the repairs and the Prompt, and the continuation after an abandoned compaction never resends the absorbed input. ## Concurrency model diff --git a/packages/docs/content/agent-loop.zh.md b/packages/docs/content/agent-loop.zh.md index 72da422..cb44c8a 100644 --- a/packages/docs/content/agent-loop.zh.md +++ b/packages/docs/content/agent-loop.zh.md @@ -24,7 +24,7 @@ session.run(newMessages, { approve, signal }) │ │ ▼ │ │ │ Environment.executeTool ──► 并发执行,输出流式回传 │ │ └─ LLMOutcome: │ -│ timeout / malformed ──► 同轮自动重连(≤5 次, │ +│ failed/timeout/malformed ──► 同轮自动重连(≤5 次, │ │ 附 [turn_retried],工具不重跑) │ │ token_usage + request_end(LLM 流结束即产出,不等工具) │ │ │ @@ -88,7 +88,7 @@ Task 运行期间,宿主可通过 `session.steer(text)` 排队一条用户消 ## 自动重连 -只有 LLM 侧的 `timeout`(网络超时、传输层断连如 socket 被对端关闭、限流、5xx,以及 `insufficient_user_quota` 这类瞬时的供应商额度/订阅错误)与 `malformed`(流截断、JSON 解析失败)会触发引擎内自动重连:同一次 `run` 内重发原始输入,并附加 `[turn_retried]` 块携带上一次的部分输出,避免工具重复执行。默认最多重连 5 次,指数退避并设上限(基数 250ms、上限 30s:250ms、500ms、1s、2s、4s,总耐心约 7.75s——所有可重试类别共用一张时间表,向较慢的类别递增,头几步仍与传输层抖动所需的一样快);超限后该轮以 `failed` 收场。每次失败的 `request_end` 会以 `retry_in_ms` 宣告计划中的等待(与实际休眠同一公式),Web App 据此实时倒计时,并提供「立即重试」(经 `Session.skipReconnectWait` 跳过剩余等待——重试计数不变)与「放弃」(普通中断;引擎的退避中中断路径结束本轮)两个内联按钮。压缩请求使用独立的更小上限(重试 3 次):压缩失败本就有优雅语义——保留原上下文、等下一次触发再试——快速失败好过让会话干等。鉴权错误在任何重试启发式之前判定、从不重试:请求以专属终态 `auth` 收场(Session 锁定的只是模型引用,凭据在会话装载时取自当前 Project 配置),Web App 据此禁用该 Session 的输入框,直到该模型的凭据被更新(更新后自动解锁)或用户点击「重试」。工具错误从不重试——它们作为 `tool_call_output` 反馈给模型,由模型决定下一步。 +除 `auth` 外,LLM 侧的所有失败都会触发引擎内自动重连——`timeout`(网络超时、传输层断连、限流、5xx、瞬时的供应商额度错误)、`malformed`(流截断、JSON 解析失败),**以及 `failed`**。`failed` 也重试,尽管分类器判定它不是瞬时错误:那个判定本质是一张允许清单(已知错误码、状态码与消息措辞),所以用自己说法描述瞬时故障的网关(例如 `Upstream HTTP/2 stream failed`)会落到这一档,此前会直接终止本轮。重试一个真正的永久错误,代价是走完退避梯度后以同样的方式收场;而把瞬时错误直接中断,则毁掉这一轮。注意改的是**策略**而非**分类**:`failed` 请求在 `request_end` 与成本中心里仍然记为 `failed`,不会被改标成超时。重连时同一次 `run` 内重发原始输入,并附加 `[turn_retried]` 块携带上一次的部分输出,避免工具重复执行。默认最多重连 5 次,指数退避并设上限(基数 250ms、上限 30s:250ms、500ms、1s、2s、4s,总耐心约 7.75s——所有可重试类别共用一张时间表,向较慢的类别递增,头几步仍与传输层抖动所需的一样快);超限后该轮以 `failed` 收场。每次失败的 `request_end` 会以 `retry_in_ms` 宣告计划中的等待(与实际休眠同一公式),Web App 据此实时倒计时,并提供「立即重试」(经 `Session.skipReconnectWait` 跳过剩余等待——重试计数不变)与「放弃」(普通中断;引擎的退避中中断路径结束本轮)两个内联按钮,CLI 则打印自己的 `[重试]` 行。三种可重试终态的渲染完全一致——用户看不见的重试,等于一次没有任何解释、也无从退出的卡顿。压缩请求重试同样的终态,只是用独立的更小上限(重试 3 次):压缩放弃后会保留原上下文、等下一次触发再试,所以那里用一段短梯度好过让会话干等完整的一段。鉴权错误在任何重试启发式之前判定、从不重试:请求以专属终态 `auth` 收场(Session 锁定的只是模型引用,凭据在会话装载时取自当前 Project 配置),Web App 据此禁用该 Session 的输入框,直到该模型的凭据被更新(更新后自动解锁)或用户点击「重试」。工具错误从不重试——它们作为 `tool_call_output` 反馈给模型,由模型决定下一步。 ## 上下文压缩(Compaction) @@ -113,7 +113,7 @@ interface CompactionSettings { 两种模式:`summarize`(默认)向旧上下文追加压缩 Prompt,提取 `[summary]` 后包装为 `[context_summary]` 用户文本,在**全新的模型上下文**中继续;`discard` 直接丢弃旧上下文。系统标记统一写作 `[tag]…[/tag]`;读取旧 Trace 与旧压缩 Prompt 时仍识别早期的尖括号形式(``、`` 等)。压缩时 [Trace 文件随之轮转](/sessions-and-traces)(`_002`、`_003`……),一个 Trace 文件恒等于一个完整模型上下文。`session.compact()` 前可用 `compactability()` 探询可行性(`ok | unsupported | empty | just_compacted`)。 -压缩请求**保持会话工具集不变**——请求前缀(含工具列表)与普通轮次逐字节一致,确保上下文最大的时刻提供商的提示词缓存依然有效。只有得到有效摘要,压缩才算成功:若响应中出现工具调用、或提取出的摘要为空,则判为无效并重试——工具调用会先以合成的失败输出逐一应答(保持 `tool_use`/`tool_result` 配对完整),修复后的请求**立即重发**(无效摘要是模型行为而非传输故障,不做退避),最多允许 5 次无效尝试,之后压缩以 `failed` 结束,保留原上下文与 Trace 文件,等待下次触发。传输层 `timeout`/`malformed` 仍走上文「自动重连」一节所述的压缩专用重连上限与退避阶梯。首个**已提交**的尝试(无论被采纳还是被判无效)同时会把折叠进压缩请求的本轮输入(任务中途的工具结果、手动 `/compact` 折叠的补发内容)吸收进旧上下文:重试只重发修复输出与压缩 Prompt,压缩放弃后的续跑也不再重发已吸收的输入。 +压缩请求**保持会话工具集不变**——请求前缀(含工具列表)与普通轮次逐字节一致,确保上下文最大的时刻提供商的提示词缓存依然有效。只有得到有效摘要,压缩才算成功:若响应中出现工具调用、或提取出的摘要为空,则判为无效并重试——工具调用会先以合成的失败输出逐一应答(保持 `tool_use`/`tool_result` 配对完整),修复后的请求**立即重发**(无效摘要是模型行为而非传输故障,不做退避),最多允许 5 次无效尝试,之后压缩以 `failed` 结束,保留原上下文与 Trace 文件,等待下次触发。可重试的结束态(`failed`/`timeout`/`malformed`,与轮次循环同一套)走上文「自动重连」一节所述的压缩专用重连上限与退避阶梯;只有 `auth` 会让压缩当场停止。首个**已提交**的尝试(无论被采纳还是被判无效)同时会把折叠进压缩请求的本轮输入(任务中途的工具结果、手动 `/compact` 折叠的补发内容)吸收进旧上下文:重试只重发修复输出与压缩 Prompt,压缩放弃后的续跑也不再重发已吸收的输入。 ## 并发模型 diff --git a/packages/docs/content/architecture.en.md b/packages/docs/content/architecture.en.md index 2bf5b8e..c8f8dba 100644 --- a/packages/docs/content/architecture.en.md +++ b/packages/docs/content/architecture.en.md @@ -129,7 +129,7 @@ The Server keeps an additional SQLite index (users, authorization, usage stats) ## Key design decisions - **One protocol, three jobs**: OmniMessage is simultaneously the SDK's external interface, the Trace on-disk format and the engine's internal currency — what streams, what is stored and what the model sees are the same thing. -- **Errors converge into messages**: the LLM and Environment never throw into the engine; results carry a six-value `stop_reason` (`completed | failed | aborted | timeout | malformed | auth`), and only LLM-side `timeout / malformed` trigger an in-run reconnect (up to 5 times with an exponential-with-ceiling backoff — `timeout` covers network timeouts, transport disconnects and transient provider quota errors; `auth` stops like `failed`). +- **Errors converge into messages**: the LLM and Environment never throw into the engine; results carry a six-value `stop_reason` (`completed | failed | aborted | timeout | malformed | auth`), and every LLM-side status except `auth` triggers an in-run reconnect (`failed / timeout / malformed`, up to 5 times with an exponential-with-ceiling backoff). `auth` is the one terminal class: a rejected credential cannot be retried into working. Retrying `failed` is a policy choice — the status itself is still reported as `failed`. - **A thin model layer**: core defines only `LLMInterface`; provider adaptation lives entirely in AgentHub (`@prismshadow/agenthub`), which is what makes any OpenAI-compatible endpoint reachable. See [Models & Providers](/models). Source entry points: `packages/core/src/engine/context-engine.ts`, `packages/core/src/interfaces.ts`. diff --git a/packages/docs/content/architecture.zh.md b/packages/docs/content/architecture.zh.md index 0d9b8b9..3a6b0ef 100644 --- a/packages/docs/content/architecture.zh.md +++ b/packages/docs/content/architecture.zh.md @@ -129,7 +129,7 @@ Server 额外维护一个 SQLite 索引库(用户、授权、用量统计),但 ## 关键设计决策 - **一个协议,三种职责**:OmniMessage 同时是 SDK 对外接口、Trace 落盘格式与引擎内部通货——「流出去的」「存下来的」「模型看到的」是同一种东西。 -- **错误收敛为消息**:LLM 与 Environment 从不向引擎抛异常;结果携带六值 `stop_reason`(`completed | failed | aborted | timeout | malformed | auth`),仅 LLM 侧的 `timeout / malformed` 触发引擎内重连(至多 5 次、指数退避设上限——`timeout` 涵盖网络超时、传输层断连与瞬时的供应商额度错误;`auth` 与 `failed` 同样直接停止)。 +- **错误收敛为消息**:LLM 与 Environment 从不向引擎抛异常;结果携带六值 `stop_reason`(`completed | failed | aborted | timeout | malformed | auth`),除 `auth` 外的所有 LLM 侧状态都会触发引擎内重连(`failed / timeout / malformed`,至多 5 次、指数退避设上限)。`auth` 是唯一的终态类别:凭据被拒绝,重试不可能让它变对。重试 `failed` 是策略选择——该状态本身仍如实上报为 `failed`。 - **薄模型层**:core 只定义 `LLMInterface`,Provider 适配全部下沉到 AgentHub(`@prismshadow/agenthub`),因此支持任意 OpenAI 兼容端点,见[模型与 Provider](/models)。 源码入口:`packages/core/src/engine/context-engine.ts`、`packages/core/src/interfaces.ts`。 diff --git a/packages/docs/content/interfaces.en.md b/packages/docs/content/interfaces.en.md index 2c96209..a92bfce 100644 --- a/packages/docs/content/interfaces.en.md +++ b/packages/docs/content/interfaces.en.md @@ -62,9 +62,9 @@ interface LLMOutcome { | `completed` | finished normally (token_usage already emitted) | proceed | | `timeout` | timeout / transport disconnect / transient provider quota error | auto-reconnect within the run | | `malformed` | response parse failure | auto-reconnect within the run | +| `failed` | an error the classifier did not judge transient (params, …) | auto-reconnect within the run as well — the status is still reported as `failed` | | `aborted` | user interrupt | stop, hand back to the user | -| `failed` | non-retryable (params, …) | stop, hand back to the user | -| `auth` | credentials rejected | stop like `failed`; hosts gate input until the model's API key is updated | +| `auth` | credentials rejected | stop, hand back to the user — the one LLM status that never retries; hosts gate input until the model's API key is updated | Implementation constraints: never throw; no internal retries — reconnecting is the engine's job (see [The Agent Loop](/agent-loop)). diff --git a/packages/docs/content/interfaces.zh.md b/packages/docs/content/interfaces.zh.md index 60754a5..3f373c9 100644 --- a/packages/docs/content/interfaces.zh.md +++ b/packages/docs/content/interfaces.zh.md @@ -62,9 +62,9 @@ interface LLMOutcome { | `completed` | 正常完成(已产出 token_usage) | 继续下一步 | | `timeout` | 超时/传输层断连/瞬时的供应商额度错误 | 同一 run 内自动重连 | | `malformed` | 响应解析失败 | 同一 run 内自动重连 | +| `failed` | 分类器未判定为瞬时的错误(参数等) | 同样在同一 run 内自动重连——状态本身仍如实上报为 `failed` | | `aborted` | 用户中断 | 停止交还用户 | -| `failed` | 参数等不可重试错误 | 停止交还用户 | -| `auth` | 凭据被拒绝 | 与 `failed` 同样停止;宿主据此禁用输入,直到该模型的 API key 被更新 | +| `auth` | 凭据被拒绝 | 停止交还用户——唯一从不重试的 LLM 终态;宿主据此禁用输入,直到该模型的 API key 被更新 | 实现约束:从不抛异常;不做内部重试(重连是引擎的职责,见 [Agent 运行循环](/agent-loop))。 diff --git a/packages/docs/content/message-flow.en.md b/packages/docs/content/message-flow.en.md index 2ae3045..3d45c47 100644 --- a/packages/docs/content/message-flow.en.md +++ b/packages/docs/content/message-flow.en.md @@ -102,7 +102,7 @@ Therefore **a renderer must never reconstruct the context from arrival order** | Case | Observable order on the stream | | --- | --- | | User interrupt | (messages produced so far) → the `abort` event — the last message before `run` returns; carry-over goes to the model context only, never streamed, never written to Trace | -| Automatic reconnect | `request_end(timeout \| malformed)` → a fresh `request_begin` (up to 5 times, exponential backoff with a 30s ceiling; `timeout` covers transport disconnects and transient provider quota errors too); the `[turn_retried]` block is model-visible only | +| Automatic reconnect | `request_end(failed \| timeout \| malformed)` → a fresh `request_begin` (up to 5 times, exponential backoff with a 30s ceiling; only `auth` skips the ladder); the `[turn_retried]` block is model-visible only | | Compaction | `compaction_begin` → the compaction request runs against the old context (its streamed output is **not** forwarded, only written to Trace) → that request's `token_usage` → `compaction_end(status)` | | max_turns reached | a length notice → the run ends; unsubmitted input is kept as carry-over | | The Prompt itself | written to Trace, not echoed back onto the stream (the caller already has it) | diff --git a/packages/docs/content/message-flow.zh.md b/packages/docs/content/message-flow.zh.md index 82cb61e..97403c9 100644 --- a/packages/docs/content/message-flow.zh.md +++ b/packages/docs/content/message-flow.zh.md @@ -101,7 +101,7 @@ Human ──run(newMessages)──► engine | 情形 | 流上可见的顺序 | | --- | --- | | 用户中断 | (已产出的消息)→ `abort` 事件——`run` 返回前的最后一条;补发内容只进模型上下文,不上流、不进 Trace | -| 自动重连 | `request_end(timeout \| malformed)` → 新的 `request_begin`(至多 5 次,指数退避、上限 30s;`timeout` 同样涵盖传输层断连与瞬时的供应商额度错误);`[turn_retried]` 块仅模型可见 | +| 自动重连 | `request_end(failed \| timeout \| malformed)` → 新的 `request_begin`(至多 5 次,指数退避、上限 30s;只有 `auth` 不走梯度);`[turn_retried]` 块仅模型可见 | | 上下文压缩 | `compaction_begin` → 压缩请求在旧上下文中执行(其流式输出**不上行**,只写 Trace)→ 该请求的 `token_usage` → `compaction_end(status)` | | 达到 max_turns | 长度提示消息 → 结束;未提交的输入按补发保留 | | Prompt 本身 | 写入 Trace,但不回流(输入方已有) | diff --git a/packages/docs/content/omni-message.en.md b/packages/docs/content/omni-message.en.md index 4ded091..0ab9eb5 100644 --- a/packages/docs/content/omni-message.en.md +++ b/packages/docs/content/omni-message.en.md @@ -258,8 +258,8 @@ type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed" | | `aborted` | user interrupt | stop, hand back to the user | | `timeout` | LLM timeout / transport disconnect / transient provider quota error | LLM side only: auto-reconnect within the run | | `malformed` | parse failure / truncated stream | LLM side only: auto-reconnect within the run | -| `failed` | other non-retryable error | stop, hand back to the user | -| `auth` | the provider rejected the credentials | stop like `failed`; hosts gate input until the model's API key is updated (credentials come from the current Project config) | +| `failed` | an error the classifier did not judge transient (LLM); a tool error (Environment) | LLM side: auto-reconnect within the run as well — the status is still reported as `failed`. Environment side: the error is fed back to the model, never retried | +| `auth` | the provider rejected the credentials | stop, hand back to the user — the one LLM status that never retries; hosts gate input until the model's API key is updated (credentials come from the current Project config) | Errors never cross an interface boundary as exceptions — they *are* messages. See [The Agent Loop](/agent-loop). diff --git a/packages/docs/content/omni-message.zh.md b/packages/docs/content/omni-message.zh.md index 82c6c8a..b2e8af0 100644 --- a/packages/docs/content/omni-message.zh.md +++ b/packages/docs/content/omni-message.zh.md @@ -256,8 +256,8 @@ type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed" | | `aborted` | 用户中断 | 停止并交还用户 | | `timeout` | LLM 超时/传输层断连/瞬时的供应商额度错误 | 仅 LLM 侧:同一 run 内自动重连 | | `malformed` | 响应解析失败/流截断 | 仅 LLM 侧:同一 run 内自动重连 | -| `failed` | 其他不可重试错误 | 停止并交还用户 | -| `auth` | 供应商拒绝了凭据 | 与 `failed` 同样停止;宿主据此禁用输入,直到该模型的 API key 被更新(凭据取自当前 Project 配置) | +| `failed` | 分类器未判定为瞬时的错误(LLM 侧);工具执行出错(Environment 侧) | LLM 侧:同样在同一 run 内自动重连——该状态本身仍如实上报为 `failed`。Environment 侧:错误回灌给模型,从不重试 | +| `auth` | 供应商拒绝了凭据 | 停止并交还用户——唯一从不重试的 LLM 终态;宿主据此禁用输入,直到该模型的 API key 被更新(凭据取自当前 Project 配置) | 错误从不以异常形式穿过接口边界——它们就是消息,见 [Agent 运行循环](/agent-loop)。 diff --git a/packages/server/src/runtime/error-recorder.ts b/packages/server/src/runtime/error-recorder.ts index 8cf3085..d1b82ed 100644 --- a/packages/server/src/runtime/error-recorder.ts +++ b/packages/server/src/runtime/error-recorder.ts @@ -8,14 +8,16 @@ * * - `expected`: anticipated by the system, has a defined handling path, part of normal * operation, no human needed — HTTP business errors (`HttpError`, mostly 4xx); LLM - * `timeout` / `malformed` (the engine already reconnects and retries); tool execution - * `failed` / `timeout` (the error is fed back to the model, and the Agent adjusts on - * its own). + * `timeout` / `malformed`, and an LLM `failed` the engine went on to retry (the engine + * reconnects on all three, so a request the ladder carried cost the user nothing); tool + * execution `failed` / `timeout` (the error is fed back to the model, and the Agent + * adjusts on its own). * - `unexpected`: shouldn't happen, usually a bug or a config/environment fault, * **needs a human** — internal errors converged to 500; process crashes; runtime * errors escaping from background tasks (Session drive / usage persistence / title - * generation / subagent registration); LLM `failed` (not retryable: auth failure, - * invalid params, etc.). + * generation / subagent registration); LLM `auth` (the credential was rejected and only + * a human can replace it) and an LLM `failed` the retries did not recover (the run ended + * on it and the user lost the turn). * - User-initiated actions **are not errors** and are never recorded: request/tool * `aborted` (user clicked "stop", or denied a tool). * diff --git a/packages/server/src/runtime/stream-error-watcher.ts b/packages/server/src/runtime/stream-error-watcher.ts index 6f1dcce..dd8b1ef 100644 --- a/packages/server/src/runtime/stream-error-watcher.ts +++ b/packages/server/src/runtime/stream-error-watcher.ts @@ -9,8 +9,19 @@ * (its state wraps up accordingly, see close). * * LLM (source = `llm`): reads the status of `request_end` — - * - `failed` / `auth` → unexpected (not retryable: credentials rejected, invalid params, - * etc., needs a human; both share the llm_failed code — no separate taxonomy); + * - `auth` → unexpected (`llm_auth`): the credential was rejected, the engine never retries + * it, and only a human holding the key can fix it. Its own code rather than the failed + * bucket, because dedup is `(source, code, Project)` over a short window: sharing a code + * with a class that now fires on every recovered blip would let a real credential failure + * be swallowed as a duplicate of something already handled. + * - `failed` → depends on whether the retry ladder carried it. The engine reconnects on + * `failed` too, so the same status covers both "a gateway hiccup nobody needed to know + * about" and "the run died on it": + * - another attempt followed (a `request_begin` resolved the pending record) → expected + * (`llm_failed_retried`), same footing as timeout/malformed: recorded for the record, + * not raised at an operator; + * - nothing followed but the end of the run (an `abort`, or close) → unexpected + * (`llm_failed`): the retries did not recover it, the user lost the turn, needs a human. * - `timeout` / `malformed` → expected (the engine already reconnects and retries, part * of normal operation); * - `aborted` / `completed` are not recorded (the former is a user-initiated interrupt, @@ -24,7 +35,8 @@ * - Immediately followed by `abort` → use its reason as the message (the real reason); * - Immediately followed by `request_begin` (the engine is retrying — no abort will ever * arrive) → use the staged request_end's own `message` (the real detail, e.g. a quota - * code), falling back to the generic status text when it carried none; + * code), falling back to the generic status text when it carried none. This boundary is + * also what tells a survived `failed` from an exhausted one (see the classification above); * - Still unresolved when the run ends → close persists it the same way. * Exception: when reason is a user-interrupt message (`aborted …`), it's not trusted — * "the user clicked stop during backoff" isn't the reason for this timeout, so the @@ -78,12 +90,31 @@ export const ORIGIN_CTX_MAX = 200; /** Recorded LLM failure states (`aborted` / `completed` are not errors and aren't included here). */ type LlmFailure = "failed" | "timeout" | "malformed" | "auth"; -/** LLM failure state → error code, classification, and fallback message (used when the abort reason isn't available). */ -const LLM_FAILURES: Record = { - failed: { code: "llm_failed", kind: "unexpected", text: "LLM request failed (not retryable)." }, - // Credentials rejection shares the failed bucket (no new taxonomy): same code/kind, the - // real reason arrives via the abort reason or the event's own message as usual. - auth: { code: "llm_failed", kind: "unexpected", text: "LLM request failed (not retryable)." }, +/** Error code, classification, and fallback message (the last used when no abort reason is available). */ +interface FailureSpec { + code: string; + kind: ErrorKind; + text: string; +} + +/** LLM failure state → its spec, for a failure the retry ladder did NOT carry (see the file header). */ +const LLM_FAILURES: Record = { + // Reached here only when no further attempt followed: the ladder ran out (or the run ended + // on it), so the user lost the turn and someone should look. A `failed` that was retried + // takes LLM_FAILED_RETRIED instead. + failed: { + code: "llm_failed", + kind: "unexpected", + text: "LLM request failed and the retries did not recover it.", + }, + // Its own code, not the failed bucket: dedup is `(source, code, Project)` over a short + // window, and `llm_failed_retried` can now fire on any recovered blip — sharing a bucket + // would let a genuine credential failure be dropped as a duplicate of one. + auth: { + code: "llm_auth", + kind: "unexpected", + text: "LLM request rejected: the provider did not accept the credentials.", + }, timeout: { code: "llm_timeout", kind: "expected", @@ -96,6 +127,18 @@ const LLM_FAILURES: Record { // —— LLM —— - it("LLM failed → unexpected (needs a human); message takes the abort reason", () => { + it("an unrecovered LLM failed → unexpected (needs a human); message takes the abort reason", () => { + // Nothing follows this failure but the abort, so the ladder did not carry it: the user + // lost the turn and it belongs in front of an operator. const got = feed([ requestBegin(), requestEnd("failed"), - abortEvent("llm request error: 401 invalid api key"), + abortEvent("llm request failed after 5 retries: 400 unknown parameter"), ]); expect(got).toHaveLength(1); expect(got[0]).toMatchObject({ source: "llm", kind: "unexpected", code: "llm_failed", - message: "llm request error: 401 invalid api key", + message: "llm request failed after 5 retries: 400 unknown parameter", project_id: "p1", agent_id: "a1", session_id: "s1", @@ -406,9 +408,10 @@ describe("stream-error-watcher (LLM / Environment errors)", () => { }); }); - it("request_end(auth) shares the llm_failed/unexpected bucket; message takes the abort reason", () => { - // Credentials rejection is its own stop reason but no new error taxonomy: same code - // and kind as failed, resolved by the abort that follows like any failed exit. + it("request_end(auth) gets its own llm_auth code, out of the failed dedup bucket", () => { + // Credentials rejection needs a code of its own now that `failed` fires on every blip + // the ladder absorbs: dedup is (source, code, Project) over a short window, so sharing + // a bucket would let a real credential failure be dropped as a duplicate. const got = feed([ requestBegin(), requestEnd("auth", "401 invalid x-api-key (invalid_api_key)"), @@ -418,11 +421,49 @@ describe("stream-error-watcher (LLM / Environment errors)", () => { expect(got[0]).toMatchObject({ source: "llm", kind: "unexpected", - code: "llm_failed", + code: "llm_auth", message: "llm request error: 401 invalid x-api-key (invalid_api_key)", }); }); + it("a failed the ladder carried → expected under its own code, not an operator incident", () => { + // The inversion this guards against: the engine retries `failed`, so the same status now + // covers "a gateway hiccup the user never saw" and "the run died on it". A following + // request_begin proves another attempt happened — that one is expected, like a timeout. + const got = feed([ + requestBegin(), + requestEnd("failed", "Upstream HTTP/2 stream failed (upstream_http2_stream_error)"), + requestBegin(), + requestEnd("completed"), + ]); + expect(got).toHaveLength(1); + expect(got[0]).toMatchObject({ + source: "llm", + kind: "expected", + code: "llm_failed_retried", + // No abort ever arrives on the retry path: the staged request_end's own detail is the + // message of record. + message: "Upstream HTTP/2 stream failed (upstream_http2_stream_error)", + }); + }); + + it("a recovered failed does not dedup away a credential failure that lands right after", () => { + // The two share a 2s dedup window and used to share the `llm_failed` code, so the auth + // record was dropped outright — the one failure that always needs a human, silenced by + // the one that never does. + const got = feed([ + requestBegin(), + requestEnd("failed", "Upstream HTTP/2 stream failed"), + requestBegin(), // The retry: resolves the failure above as recovered. + requestEnd("auth", "401 invalid x-api-key"), + abortEvent("llm request error: 401 invalid x-api-key"), + ]); + expect(got.map((r) => [r.code, r.kind])).toEqual([ + ["llm_failed_retried", "expected"], + ["llm_auth", "unexpected"], + ]); + }); + it("a retried failure keeps its real detail: request_end(timeout).message lands in the record", () => { // The retry path: the engine reconnects (request_begin) and eventually succeeds, so no // abort ever arrives for the staged failure — the request_end's own failure detail @@ -498,8 +539,9 @@ describe("stream-error-watcher (LLM / Environment errors)", () => { w.close(); const got = rows(); expect(got).toHaveLength(1); + // close() is not proof of a retry, so it takes the conservative branch: unrecovered. expect(got[0]).toMatchObject({ code: "llm_failed", kind: "unexpected" }); - expect(got[0]!.message).toContain("LLM request failed"); + expect(got[0]!.message).toBe("LLM request failed and the retries did not recover it."); }); it("parent/child LLM failures pend separately by origin; abort reasons never cross over", () => { diff --git a/packages/server/test/session-manager.test.ts b/packages/server/test/session-manager.test.ts index 917d44c..bc69264 100644 --- a/packages/server/test/session-manager.test.ts +++ b/packages/server/test/session-manager.test.ts @@ -214,7 +214,7 @@ describe("session-manager", () => { expect(captured.map((a) => [a.source, a.code, a.kind])).toEqual([ ["environment", "tool_failed:write_file", "expected"], // error fed back to the model; the Agent adjusts on its own - ["llm", "llm_failed", "unexpected"], // not retryable, requires human intervention + ["llm", "llm_failed", "unexpected"], // the abort follows it: the retries did not recover it, so a human is needed ]); expect(captured[0]!.ctx).toEqual({ projectId: "p1", agentId: "a1", sessionId: "session-1" }); expect(String(captured[0]!.err)).toContain("[tool error] exit code 2"); diff --git a/packages/server/test/trace-service.test.ts b/packages/server/test/trace-service.test.ts index 0ed2d6a..a596d0d 100644 --- a/packages/server/test/trace-service.test.ts +++ b/packages/server/test/trace-service.test.ts @@ -378,6 +378,52 @@ describe("trace-service", () => { expect(a.elapsedMs).toBe(4_100); }); + it("a failed request the engine retries stays inside the same turn (it is a reconnect, not a turn end)", async () => { + // The engine reconnects on `failed` as well as timeout/malformed, so the resent Request + // belongs to the same user turn. If segmentation still keyed on timeout/malformed only, + // one gateway blip would split a single turn's Tokens, duration and TPS across two Tasks + // and inflate the Task count — the same smearing the timeout rule exists to prevent. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + at("2026-07-05T10:00:00.000Z", sessionMeta(metaPayload())), + at("2026-07-05T10:00:00.000Z", userText("question one")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:02.000Z", requestEnd("failed")), // a gateway error → the engine reconnects + at("2026-07-05T10:00:03.000Z", requestBegin()), + at("2026-07-05T10:00:05.000Z", requestEnd("completed")), + at("2026-07-05T10:00:05.100Z", tokenUsage(counts(1000), buckets(0, 900, 100))), + ]); + const a = await service.analyze(P, A, S, 1); + expect(a.requests.map((r) => r.taskIndex)).toEqual([0, 0]); // one turn, two Requests + expect(a.tasks.map((t) => t.taskIndex)).toEqual([0]); + expect(a.reconnectCount).toBe(1); + expect(a.tasks[0]!.tokens.output).toBe(100); + }); + + it("a failed compaction request is NOT a reconnect: compaction fails fast on it", async () => { + // Segmentation has to track the engine loop by loop: the turn loop retries `failed`, the + // compaction loop deliberately does not (it keeps the old context and tries again on the + // next trigger). Counting this as a reconnect would invent an attempt that never happened. + await writeTraceFile(root, P, A, "2026-07-05", S, 1, [ + at("2026-07-05T10:00:00.000Z", sessionMeta(metaPayload())), + at("2026-07-05T10:00:00.000Z", userText("question one")), + at("2026-07-05T10:00:01.000Z", requestBegin()), + at("2026-07-05T10:00:02.000Z", requestEnd("completed")), + at("2026-07-05T10:00:02.100Z", tokenUsage(counts(1000), buckets(0, 900, 100))), + at( + "2026-07-05T10:00:03.000Z", + compactionBegin({ reason: "context", mode: "summarize", context: 1000, turns: 1 }), + ), + at("2026-07-05T10:00:04.000Z", requestBegin()), + at("2026-07-05T10:00:05.000Z", requestEnd("failed")), // compaction gives up here + at( + "2026-07-05T10:00:06.000Z", + compactionEnd({ reason: "context", mode: "summarize", status: "failed" }), + ), + ]); + const a = await service.analyze(P, A, S, 1); + expect(a.reconnectCount).toBe(0); + }); + it("after the previous turn ends in timeout (retries exhausted), a new user message starts a new turn", async () => { // "timeout → continuation" holds only for **automatic retries within the // same run**. Once retries are exhausted and the engine gives up, a message diff --git a/packages/web/src/lib/omni/stream-model.ts b/packages/web/src/lib/omni/stream-model.ts index 1b0cf72..6622595 100644 --- a/packages/web/src/lib/omni/stream-model.ts +++ b/packages/web/src/lib/omni/stream-model.ts @@ -22,8 +22,8 @@ * token_usage counts toward this level's stats (same convention as the CLI). * - Events: approval_decision annotates the corresponding tool card * (labeled "manual" if clicked on this end, "automatic" otherwise); - * abort → an interruption marker item; request_end ending in - * timeout/malformed → a retry-hint item (the engine discards that + * abort → an interruption marker item; request_end ending in any status + * the engine reconnects on (failed/timeout/malformed) → a retry-hint item (the engine discards that * attempt and resends the original input; the next request_begin marks the hint as resent, and an * arriving abort marks it as retries exhausted); other request_begin/end * events aren't rendered (Request duration is covered by Trace @@ -195,12 +195,23 @@ export interface AbortItem { reason?: string; } -/** An LLM Request ending in timeout/malformed → the engine retries carrying the content already produced. */ +/** + * The statuses the engine reconnects on — every LLM failure except `auth`, which is terminal + * (see core's TURN_RETRY_STATUSES). A retry the user cannot see is a stalled session with no + * explanation and no way out, so all three render the same countdown and the same controls. + */ +export type ReconnectStatus = "failed" | "timeout" | "malformed"; + +function isReconnectStatus(status: StopReason | undefined): status is ReconnectStatus { + return status === "failed" || status === "timeout" || status === "malformed"; +} + +/** An LLM Request ending in failed/timeout/malformed → the engine retries carrying the content already produced. */ export interface ReconnectItem { kind: "reconnect"; id: number; - /** Trigger reason: timeout (timed out / disconnected) or malformed (an incomplete or unparseable response). */ - status: "timeout" | "malformed"; + /** Trigger reason: timeout (timed out / disconnected), malformed (an incomplete or unparseable response), or failed (the provider returned an error). */ + status: ReconnectStatus; /** Which retry attempt this is (increments on consecutive failures within the same round; resets to 1 after a request finishes normally). */ attempt: number; /** The retry request has been sent (set true by the next request_begin). */ @@ -349,7 +360,7 @@ export interface StreamModel { * dispatches it via `void executeOne`, which doesn't block the streaming loop — execution happens between two Requests). */ openApprovalWaitMs: number; - /** Consecutive reconnect-failure count (incremented when request_end is timeout/malformed, reset to zero on any other terminal status). */ + /** Consecutive reconnect-failure count (incremented when request_end carries a status the engine reconnects on, reset to zero on any other terminal status). */ reconnectRun: number; /** * Timestamp (ms) of the most recent auth failure: a main-session `request_end` with @@ -1262,9 +1273,9 @@ function handleEvent(model: StreamModel, p: EventPayload, tsMs?: number, nowMs?: return; } case "request_end": { - // timeout/malformed: the engine retries carrying the content already - // produced, rendering a retry hint (with the attempt number); other - // terminal statuses aren't rendered (Request duration is covered by Trace performance + // failed/timeout/malformed: the engine retries carrying the content already + // produced, rendering a retry hint (with the attempt number); the terminal + // statuses aren't rendered (Request duration is covered by Trace performance // analysis) and reset the consecutive-failure count. request events // within a compaction range (only visible during history rebuild) // are neither rendered nor counted — the compaction process only exposes the compaction event pair to the Human. @@ -1302,7 +1313,11 @@ function handleEvent(model: StreamModel, p: EventPayload, tsMs?: number, nowMs?: // round, so the pending compaction usage never reaches this step and is discarded at finalization (not counted into this round). if (tsMs !== undefined) model.taskLastReqEndMs = tsMs; commitPendingCompaction(model.stats); - if (p.status === "timeout" || p.status === "malformed") { + // Every status the engine reconnects on gets an item, `failed` included: it is retried + // exactly like the other two, so leaving it out would stall the session for the whole + // ladder with nothing on screen and no give-up control. It would also reset the counter + // mid-ladder, renumbering a mixed timeout → failed → timeout run back to retry #1. + if (isReconnectStatus(p.status)) { model.reconnectRun += 1; const item: ReconnectItem = { kind: "reconnect", diff --git a/packages/web/src/lib/strings-en.ts b/packages/web/src/lib/strings-en.ts index 8300c1b..db7e58b 100644 --- a/packages/web/src/lib/strings-en.ts +++ b/packages/web/src/lib/strings-en.ts @@ -671,15 +671,23 @@ When done, open index.html in a browser and self-test once.`, modelAuthDeadRetry: "Retry", modelAuthDeadCta: "New Session", modelAuthDeadPlaceholder: "Model authentication failed — update the API key first", - /** Reconnect hint line; `secondsLeft` (waiting state only) switches to the live-countdown wording. */ + /** + * Reconnect hint line; `secondsLeft` (waiting state only) switches to the live-countdown + * wording. `failed` is in the union because the engine retries it like the other two — + * its cause names the provider rather than the transport, since that is where it came from. + */ reconnect: ( - status: "timeout" | "malformed", + status: "failed" | "timeout" | "malformed", state: "waiting" | "retried" | "gaveUp", attempt: number, secondsLeft?: number, ) => { const cause = - status === "timeout" ? "Connection timed out" : "Response incomplete or unparseable"; + status === "timeout" + ? "Connection timed out" + : status === "malformed" + ? "Response incomplete or unparseable" + : "The model provider returned an error"; const action = state === "gaveUp" ? "no further retries" diff --git a/packages/web/src/lib/strings.ts b/packages/web/src/lib/strings.ts index da39b3a..113fd52 100644 --- a/packages/web/src/lib/strings.ts +++ b/packages/web/src/lib/strings.ts @@ -654,14 +654,23 @@ Penguin 视觉风格(见 web-design 技能),深色/浅色主题( { - const cause = status === "timeout" ? "连接超时或网络中断" : "响应不完整或无法解析"; + const cause = + status === "timeout" + ? "连接超时或网络中断" + : status === "malformed" + ? "响应不完整或无法解析" + : "模型服务返回错误"; const action = state === "gaveUp" ? "已停止重试" diff --git a/packages/web/test/stream-model.test.ts b/packages/web/test/stream-model.test.ts index 2d08c11..23927d4 100644 --- a/packages/web/test/stream-model.test.ts +++ b/packages/web/test/stream-model.test.ts @@ -494,6 +494,67 @@ describe("approvals and events", () => { expect((items(m)[2] as ReconnectItem).attempt).toBe(1); }); + it("request_end(failed) renders a retry notice too, with its countdown inputs and give-up target", () => { + // The engine reconnects on `failed` exactly like timeout/malformed. Without an item there + // is no countdown and findLastWaitingReconnect returns null, so "Retry now" / "Give up" + // never render either — the session just stalls for up to 7.75s with nothing on screen. + const m = createStreamModel(); + pushMessage(m, requestBegin()); + pushMessage( + m, + requestEnd("failed", "Upstream HTTP/2 stream failed (upstream_http2_stream_error)", 4000), + 111_000, + ); + const retry = items(m)[0] as ReconnectItem; + expect(retry).toMatchObject({ + kind: "reconnect", + status: "failed", + attempt: 1, + retrying: false, + plannedDelayMs: 4000, // the countdown + arrivedAtMs: 111_000, // its client-clock anchor + }); + // Waiting, so it is the item the retry-now / give-up controls attach to; the retry then + // flips it out of the waiting state exactly like the other two statuses. + pushMessage(m, requestBegin()); + expect(retry.retrying).toBe(true); + }); + + it("a mixed ladder keeps counting: failed no longer resets the attempt number mid-run", () => { + // `failed` used to fall through to the reset branch, so timeout → failed → timeout + // renumbered the third attempt back to #1 while the engine was on its third. + const m = createStreamModel(); + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("timeout")); + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("failed", "502 bad gateway")); + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("timeout")); + expect((items(m) as ReconnectItem[]).map((i) => [i.status, i.attempt])).toEqual([ + ["timeout", 1], + ["failed", 2], + ["timeout", 3], + ]); + // A normal finish still resets it: the next run's first failure is #1 again. + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("completed")); + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("failed")); + expect((items(m)[3] as ReconnectItem).attempt).toBe(1); + }); + + it("request_end(auth) stays out of the ladder: terminal, so no retry notice and the count resets", () => { + // The one status the engine does not retry — an item would promise a countdown and a + // "Retry now" that will never happen, on top of the composer already being gated. + const m = createStreamModel(); + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("timeout")); + pushMessage(m, requestBegin()); + pushMessage(m, requestEnd("auth", "401 invalid x-api-key")); + expect((items(m) as ReconnectItem[]).filter((i) => i.kind === "reconnect")).toHaveLength(1); + expect(m.reconnectRun).toBe(0); + }); + it("retries exhausted: an arriving abort marks the waiting retry notice gaveUp and resets the consecutive-failure count", () => { const m = createStreamModel(); pushMessage(m, requestBegin());