diff --git a/packages/core/src/engine/context-engine.ts b/packages/core/src/engine/context-engine.ts index e1f264f..46a8f6a 100644 --- a/packages/core/src/engine/context-engine.ts +++ b/packages/core/src/engine/context-engine.ts @@ -105,6 +105,20 @@ export interface CompactionSettings { interface CompactionResult { status: StopReason; summary?: OmniMessage; + /** + * Whether at least one summarize attempt was **committed** by AgentHub (only a `completed` + * attempt commits — timeout/malformed end an incomplete stream and failed/auth/aborted throw + * or cut off before a clean end). The carry rule at every caller is a two-case binary on + * this flag (issue #85): committed → the input the caller folded in (mid-Task tool outputs, + * or the carry-over a manual `compact()` folds in) now lives in the old LLM object's history + * and must never be resent — strict providers reject the duplicates as stale tool_results; + * not committed → the folded input is untouched and is resent exactly as before. When + * nothing was folded in (idle/boundary compaction), the committed branch is vacuous — + * dropping zero outputs, clearing an empty carry — so no separate "was anything absorbed" + * signal is needed. Zero committed attempts also implies zero synthesized repairs (repairs + * only answer a committed rejection's tool calls). + */ + committed: boolean; } /** @@ -556,6 +570,9 @@ export class ContextEngine { // applies mid-Task — when runTurn returns, all of this turn's // tool results are ready and paired with their tool_call. const midTask = turn.toolOutputs.length > 0; + // Outputs this turn still owes the model: dropped when a committed compaction attempt + // consumes them into history (the two-case carry rule below, issue #85). + let turnOutputs = turn.toolOutputs; const compactionReason = this.compactionTrigger(); if (compactionReason) { const mode = this.deps.compaction!.mode; @@ -574,11 +591,20 @@ export class ContextEngine { signal, ); if (result.status === "aborted") { - // User interrupted compaction: keep the original context; if mid-Task, hold the - // tool outputs as carry-over per case A. Abort is the one path that discards the - // steering queue (run's finally — control goes back to the user). + // User interrupted compaction: keep the original context. The carry rule is the + // same two-case binary as everywhere (issue #85): a committed attempt consumed + // this turn's outputs into history — only the repair stash summarizeContext left + // in pendingCarryOver still needs resending; otherwise the outputs are untouched + // and are appended behind the stash as case-A carry-over. Abort is the one path + // that discards the steering queue (run's finally — control goes back to the + // user). if (midTask) { - this.pendingCarryOver = this.buildCarryOver(attemptInput, turn); + if (!result.committed) { + this.pendingCarryOver = [ + ...this.pendingCarryOver, + ...this.buildCarryOver(attemptInput, turn), + ]; + } yield* this.emitAbort("aborted during compaction"); } return; @@ -605,7 +631,13 @@ export class ContextEngine { continue; } // failed: keep the original context and Trace index; the current Task continues and - // retries on the next trigger (no fallback to discard). + // retries on the next trigger (no fallback to discard). The carry rule (issue #85): + if (result.committed) { + // A committed attempt consumed this turn's outputs into history — drop them from + // the continuation; only the repair stash remains pending. + turnOutputs = []; + } + // else: nothing committed — the outputs are untouched and resent below as always. } } @@ -615,9 +647,18 @@ export class ContextEngine { // ending the Task — subject to the max-turns guard at the top of the loop). const steering = yield* this.deliverSteering(); // No tool_call this turn and no steering left -> the Task ends (the final reply has - // already been streamed out). + // already been streamed out). A compaction stash, if any, rides the next run. if (!midTask && steering.length === 0) return; - nextInput = [...turn.toolOutputs, ...steering]; + // Anything a failed compaction stashed mid-run (synthesized repair outputs from + // rejected attempts) rides the very next request, ahead of the turn outputs so + // tool_results stay contiguous and first. + const stashed = this.pendingCarryOver; + this.pendingCarryOver = []; + nextInput = [...stashed, ...turnOutputs, ...steering]; + // Mid-task, but a committed compaction consumed the outputs and nothing else remains + // to send: the run ends here — the failure was surfaced via compaction_end(failed), + // the context is intact, and the next prompt continues from the committed state. + if (nextInput.length === 0) return; } } @@ -670,15 +711,27 @@ export class ContextEngine { yield* this.discardContext("manual"); return; } - const result = yield* this.summarizeContext( - "manual", - this.pendingCarryOver.map(downgradeCarriedGoalInput), - opts?.signal, - ); + // The carry seam is a clean binary on whether the compaction committed anything to + // AgentHub (PR #87 review): + // - nothing committed (every attempt timeout/malformed/failed/auth/aborted): the fold + // never reached the model context — restore the prior carry-over **verbatim**. Zero + // committed attempts also means zero synthesized repairs, so there is no stash to + // interleave with (pinned by tests); + // - something committed: the carry-over is **consumed** — it lives in the committed + // history now and must never be resent; only the repair stash (unanswered tool_call + // pairing left by a final rejection, already in pendingCarryOver) remains pending. + // Dead-goal rounds are downgraded on the drained snapshot (goal mode's consumer-site + // rule): a no-commit restore keeps the downgraded copies — the downgrade is idempotent + // and every consumer applies it anyway, while non-goal messages keep their identity. + const folded = this.pendingCarryOver.map(downgradeCarriedGoalInput); + this.pendingCarryOver = []; + const result = yield* this.summarizeContext("manual", folded, opts?.signal); if (result.status === "completed") { - this.pendingCarryOver = []; this.pendingSummary = result.summary!; + } else if (!result.committed) { + this.pendingCarryOver = folded; } + // committed but not completed: the carry-over is deliberately not restored. } /** @@ -1078,7 +1131,10 @@ export class ContextEngine { * then the compaction fails. timeout/malformed reconnect via the existing retry mechanism * under the compaction-specific cap (`compactionMaxReconnects`, tighter than the turn loop's * ladder), collapsing to failed once retries are exhausted; on failure/abort, the original - * context and Trace index are kept — it does not fall back to discard. + * context and Trace index are kept — it does not fall back to discard. The first **committed** + * attempt absorbs `pendingToolOutputs` into the old context's history (issue #85): later + * resends carry only the repairs and the Prompt, and the result's `committed` flag tells + * the caller the folded input must not be resent even though the compaction did not complete. * Docs: /docs/agent-loop § "Compaction". */ private async *summarizeContext( @@ -1095,10 +1151,18 @@ export class ContextEngine { // executed and aren't recorded again, while carry-over's not-yet-written synthetic content // (flatten text, backfilled placeholders) and the compaction Prompt are written now. const prompt = userText(settings.prompt); - const baseInput = [...pendingToolOutputs, prompt]; - let input = baseInput; + // The resend base: shrinks to the Prompt alone once an attempt commits — the folded turn + // input then lives in the old LLM object's history, and resending it would make strict + // providers reject the request over duplicate/stale tool_results (issue #85). + let base = [...pendingToolOutputs, prompt]; + let input = base; await this.write(prompt); + // Whether any attempt was committed by AgentHub (only `completed` commits: timeout and + // malformed end an incomplete stream, and failed/auth/aborted throw or cut off before a + // clean end — none of those reach the stateful commit). Returned as `committed`: the + // callers' two-case carry rule branches on it. + let committed = false; // Synthesized outputs answering the latest rejected attempt's tool calls, not yet carried // by a committed request: prepended to the retry input, and stashed as carry-over should // the compaction be abandoned first (see stashRepairs). @@ -1114,13 +1178,16 @@ export class ContextEngine { if (signal?.aborted) { this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "aborted"); - return { status: "aborted" }; + return { status: "aborted", committed }; } const attempt = await this.runCompactionRequest(input, signal, reconnects); if (attempt.status === "completed") { // The attempt was committed by AgentHub, so whatever its input carried — including // repairs synthesized for a previous rejection — is now in history and must not be - // resent. + // resent. The first commit absorbs the folded turn input: the base shrinks to the + // Prompt alone. + committed = true; + base = [prompt]; pendingRepairs = []; // A completed response counts as a compaction success only when it is a **usable // summary**: the extracted text is non-empty and the response called no tool. The @@ -1143,7 +1210,7 @@ export class ContextEngine { const summary = userText(buildContextSummaryText(summaryText)); yield* this.emitCompactionEnd(reason, "summarize", "completed"); await this.startNewContext(); - return { status: "completed", summary }; + return { status: "completed", summary, committed }; } // Rejected. Tool calls were never dispatched, yet the assistant turn holding them IS // committed on the live LLM object — leaving them unanswered would get every @@ -1163,13 +1230,14 @@ export class ContextEngine { }), ); for (const repair of pendingRepairs) await this.write(repair); - // Rebuild from baseInput rather than appending: everything the rejected attempt's - // input carried is committed, so only the fresh repairs and the Prompt go out again. - input = pendingRepairs.length > 0 ? [...pendingRepairs, ...baseInput] : baseInput; + // Rebuild from the (shrunken) base rather than appending: everything the rejected + // attempt's input carried is committed, so only the fresh repairs and the Prompt go + // out again. + input = pendingRepairs.length > 0 ? [...pendingRepairs, ...base] : base; if (rejections >= MAX_SUMMARY_REJECTIONS) { this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "failed"); - return { status: "failed" }; + return { status: "failed", committed }; } // A rejection is model behavior, not a transport failure: the request pipeline is // healthy, so the repaired input is resent immediately — no backoff and no @@ -1181,7 +1249,7 @@ export class ContextEngine { if (attempt.status === "aborted") { this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "aborted"); - return { status: "aborted" }; + return { status: "aborted", committed }; } if (attempt.status === "failed" || attempt.status === "auth") { // `auth` folds into `failed` here: the compaction event pair keeps its @@ -1190,7 +1258,7 @@ export class ContextEngine { // request will surface it; the compaction request_end is Trace-only). this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "failed"); - return { status: "failed" }; + return { status: "failed", committed }; } // timeout / malformed: retried via reconnect — transport-level, never committed by // AgentHub (case B), so the input (any pending repairs included) is resent unchanged. @@ -1200,14 +1268,14 @@ export class ContextEngine { if (reconnects >= this.compactionMaxReconnects) { this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "failed"); - return { status: "failed" }; + return { status: "failed", committed }; } reconnects += 1; const ok = await this.backoff(reconnects, signal); if (!ok) { this.stashRepairs(pendingRepairs); yield* this.emitCompactionEnd(reason, "summarize", "aborted"); - return { status: "aborted" }; + return { status: "aborted", committed }; } } } diff --git a/packages/core/test/compaction.test.ts b/packages/core/test/compaction.test.ts index 854e6bb..13096df 100644 --- a/packages/core/test/compaction.test.ts +++ b/packages/core/test/compaction.test.ts @@ -819,6 +819,209 @@ describe("context compaction", () => { expect(closureRepairs).toEqual([]); }); + it("mid-task: a committed rejection absorbs the turn's tool outputs — retries never resend them, and an empty-handed abandonment ends the run", async () => { + // Issue #85, strict-provider scenario: the first committed attempt puts the turn's tool + // outputs (folded into the compaction input) into the old LLM object's history. From then + // on the retries must carry only the repairs and the Prompt — resending the outputs would + // be rejected as stale tool_results — and after the compaction is abandoned the + // continuation must not resend them either. Here the final rejection is empty (no repairs + // left to deliver) and no steering is queued, so the run ends at the failure: the context + // is intact and the next prompt continues from the committed state. + const llm1 = new ScriptedLLM( + [ + // Turn 1: a tool call, over the threshold -> mid-task compaction. + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "ct" }), usage(150, 150)], + }, + // Attempt 1: committed but rejected (tool call) -> absorbs [output(ct), prompt]. + { + messages: [ + assistantText("[summary]nope[/summary]"), + toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), + ], + }, + // Attempts 2-5: committed but empty -> rejected; the base has shrunk to the Prompt. + { messages: [thinkingMessage("blank 2")] }, + { messages: [thinkingMessage("blank 3")] }, + { messages: [thinkingMessage("blank 4")] }, + { messages: [thinkingMessage("blank 5")] }, + // Next run after the abandonment: served by the same instance, plain prompt. + { messages: [assistantText("continuing"), usage(60, 900)] }, + ], + "llm1", + ); + let created = 0; + const trace = new Writer({ tracesDir: traces, sessionId: "sess_absorb" }); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + trace, + sessionMeta: metaMessage, + compaction: settings(), + createLLM: () => { + created += 1; + return new ScriptedLLM([], "llm2"); + }, + reconnectBackoffMs: 1, + }); + const oldPath = trace.currentPath(); + + const out1 = await collect(engine.run([userText("task one")], { approve: allowAll })); + + expect(compactionEvents(out1)[1]).toMatchObject({ type: "compaction_end", status: "failed" }); + // The run ends at the empty-handed abandonment without an abort: the compaction failure + // is the whole story. + expect(payloadTypes(out1)).not.toContain("abort"); + expect(llm1.calls).toHaveLength(6); + // Attempt 1 folds the turn's tool output in; attempt 2 carries the repair + Prompt but + // NOT the absorbed output; attempts 3-5 are Prompt-only. + expect(payloadTypes(llm1.calls[1]!)).toEqual(["tool_call_output", "text"]); + expect((llm1.calls[1]![0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("ct"); + expect(payloadTypes(llm1.calls[2]!)).toEqual(["tool_call_output", "text"]); + expect((llm1.calls[2]![0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("c1"); + expect(textOf(llm1.calls[2]![1]!)).toBe("COMPACT NOW"); + for (let attempt = 3; attempt <= 5; attempt += 1) { + expect(llm1.calls[attempt]!.map(textOf)).toEqual(["COMPACT NOW"]); + } + // Original context kept: no LLM swap, no Trace rotation. + expect(created).toBe(0); + expect(trace.currentPath()).toBe(oldPath); + + // The next run continues from the committed state: the absorbed outputs are not resent + // and nothing was stashed (the final rejection was empty). + await collect(engine.run([userText("task two")], { approve: allowAll })); + expect(llm1.calls[6]!.map(textOf)).toEqual(["task two"]); + }); + + it("mid-task: abandonment with a trailing tool-calling rejection continues the task with the repair alone", async () => { + // Any committed attempt absorbs — here the FIRST (empty) rejection commits the turn + // outputs, and the FINAL rejection leaves unanswered tool calls. The continuation after + // the failure delivers exactly the stashed repairs (never the absorbed outputs), keeping + // the live object's history well-formed while the task keeps running. + const llm1 = new ScriptedLLM( + [ + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "ct" }), usage(150, 150)], + }, + // Attempts 1-4: empty rejections (attempt 1 commits and absorbs the outputs). + { messages: [thinkingMessage("blank 1")] }, + { messages: [thinkingMessage("blank 2")] }, + { messages: [thinkingMessage("blank 3")] }, + { messages: [thinkingMessage("blank 4")] }, + // Attempt 5: rejected with a tool call -> its repair is stashed at the abandonment. + { messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c5" })] }, + // Continuation: the repair alone; the model wraps the task up (under the threshold). + { messages: [assistantText("recovered"), usage(60, 900)] }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => new ScriptedLLM([], "llm2"), + reconnectBackoffMs: 1, + }); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + expect(compactionEvents(out)[1]).toMatchObject({ type: "compaction_end", status: "failed" }); + expect(llm1.calls).toHaveLength(7); + // Attempt 1 folds the outputs; attempts 2-5 are Prompt-only (absorbed by the first commit). + expect(payloadTypes(llm1.calls[1]!)).toEqual(["tool_call_output", "text"]); + for (let attempt = 2; attempt <= 5; attempt += 1) { + expect(llm1.calls[attempt]!.map(textOf)).toEqual(["COMPACT NOW"]); + } + // The continuation input is the repair answering c5 — nothing else: no absorbed outputs, + // no prompt. + expect(payloadTypes(llm1.calls[6]!)).toEqual(["tool_call_output"]); + const cont = llm1.calls[6]![0]!.payload as { tool_call_id: string; output: string }; + expect(cont.tool_call_id).toBe("c5"); + expect(cont.output).toBe( + "[tool error] the compaction request expects a summary, not tool calls", + ); + // The task finished on the continuation; the stash is spent for later runs. + expect(out.map((m) => (m.payload as { text?: string }).text)).toContain("recovered"); + await collect(engine.run([userText("again")], { approve: allowAll })); + expect(llm1.calls[7]!.map(textOf)).toEqual(["again"]); + }); + + it("mid-task: an all-transport abandonment absorbs nothing — the outputs are resent as before", async () => { + // Counter-case: timeout/malformed attempts never commit, so the turn's tool outputs were + // never absorbed and the post-failure continuation must resend them exactly as before + // (their tool_use pairing is still unanswered on the live object). + const llm1 = new ScriptedLLM( + [ + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "ct" }), usage(150, 150)], + }, + { messages: [], outcome: { status: "timeout" } }, + { messages: [], outcome: { status: "timeout" } }, + // Continuation: the resent tool output; the model wraps up under the threshold. + { messages: [assistantText("done on old context"), usage(60, 400)] }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => new ScriptedLLM([], "llm2"), + compactionMaxReconnects: 1, + reconnectBackoffMs: 1, + }); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + expect(compactionEvents(out)[1]).toMatchObject({ type: "compaction_end", status: "failed" }); + expect(llm1.calls).toHaveLength(4); + expect(payloadTypes(llm1.calls[3]!)).toEqual(["tool_call_output"]); + expect((llm1.calls[3]![0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("ct"); + }); + + it("mid-task: an abort after a committed rejection merges the stash — repairs ride, absorbed outputs do not", async () => { + // The abort-in-window case (issue #85): the stash must merge, not be overwritten. After a + // committed rejection the turn outputs are absorbed, so the case-A carry-over is skipped + // entirely and only the unanswered repair rides the next run. + const llm1 = new ScriptedLLM( + [ + { + messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "ct" }), usage(150, 150)], + }, + // Attempt 1: committed rejection with a tool call -> absorbs outputs, repair pending. + { messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" })] }, + // Attempt 2 (carrying the repair): the user aborts mid-request -> uncommitted, the + // repair is still unanswered and gets stashed. + { messages: [], outcome: { status: "aborted" } }, + // Next run: the stash leads, then the new prompt. + { messages: [assistantText("back"), usage(60, 500)] }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => new ScriptedLLM([], "llm2"), + reconnectBackoffMs: 1, + }); + + const out = await collect(engine.run([userText("go")], { approve: allowAll })); + const events = compactionEvents(out); + expect(events[1]).toMatchObject({ type: "compaction_end", status: "aborted" }); + expect(payloadTypes(out)).toContain("abort"); + expect(llm1.calls).toHaveLength(3); + // Attempt 2 carried the repair + Prompt (not the absorbed outputs). + expect(payloadTypes(llm1.calls[2]!)).toEqual(["tool_call_output", "text"]); + expect((llm1.calls[2]![0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("c1"); + + // The next run leads with the still-unanswered repair; the absorbed turn outputs are + // NOT re-held as carry-over (no duplicate ct output). + await collect(engine.run([userText("next")], { approve: allowAll })); + const nextRun = llm1.calls[3]!; + expect(payloadTypes(nextRun)).toEqual(["tool_call_output", "text"]); + expect((nextRun[0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("c1"); + expect(textOf(nextRun[1]!)).toBe("next"); + }); + it("the compaction request carries exactly the same tools as ordinary requests (prompt-cache pin)", async () => { // Owner constraint (#84): compaction runs exactly when the context is at its largest, and // the provider's prompt cache only holds if the request prefix — the tool list included — @@ -1048,6 +1251,104 @@ describe("context compaction", () => { ]); }); + it("manual compaction that commits nothing restores the prior carry-over verbatim", async () => { + // The carry seam's binary (PR #87 review): every attempt was a transport failure, so + // nothing reached AgentHub — the folded carry-over comes back exactly as it was (the very + // same message objects, not routed through any absorption logic), and with zero committed + // attempts there are no repairs to interleave with. + const llm1 = new ScriptedLLM( + [ + { messages: [assistantText("hi"), usage(10, 10)] }, + // Manual compaction attempts: transport failures only — nothing committed. + { messages: [], outcome: { status: "timeout" } }, + { messages: [], outcome: { status: "timeout" } }, + // Run 3: the restored carry-over leads the input, exactly as before the compact(). + { messages: [assistantText("resumed"), usage(20, 40)] }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => new ScriptedLLM([], "llm2"), + compactionMaxReconnects: 1, + reconnectBackoffMs: 1, + }); + await collect(engine.run([userText("start")], { approve: allowAll })); + // An aborted-before-issue run leaves its input as carry-over. + const aborted = new AbortController(); + aborted.abort(); + const pendingMsg = userText("pending question"); + await collect(engine.run([pendingMsg], { signal: aborted.signal, approve: allowAll })); + expect(llm1.calls).toHaveLength(1); // the aborted run never reached the LLM + + const out = await collect(engine.compact()); + expect(compactionEvents(out)[1]).toMatchObject({ type: "compaction_end", status: "failed" }); + // The compaction folded the carry-over into its (uncommitted) attempts... + expect(llm1.calls).toHaveLength(3); + expect(llm1.calls[1]!.map(textOf)).toEqual(["pending question", "COMPACT NOW"]); + expect(llm1.calls[2]!.map(textOf)).toEqual(["pending question", "COMPACT NOW"]); + + // ...and restored it verbatim: the next run resends the very same message object first. + await collect(engine.run([userText("run3")], { approve: allowAll })); + expect(llm1.calls[3]![0]).toBe(pendingMsg); + expect(llm1.calls[3]!.map(textOf)).toEqual(["pending question", "run3"]); + }); + + it("manual compaction with a committed attempt consumes the carry-over; only repairs remain pending", async () => { + // The other side of the binary: the first committed attempt put the folded carry-over + // into AgentHub history, so it is disposed of at the seam — never restored — and the next + // input reflects the committed state plus the repair stash left by the final rejection. + const llm1 = new ScriptedLLM( + [ + { messages: [assistantText("hi"), usage(10, 10)] }, + // Attempt 1: committed but empty -> absorbs the folded carry-over into history. + { messages: [thinkingMessage("blank 1")] }, + // Attempts 2-4: still empty; the base has shrunk to the Prompt. + { messages: [thinkingMessage("blank 2")] }, + { messages: [thinkingMessage("blank 3")] }, + { messages: [thinkingMessage("blank 4")] }, + // Attempt 5: rejected with a tool call -> its repair is stashed at the abandonment. + { messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c9" })] }, + // Run 3: the repair leads; the consumed carry-over is NOT resent. + { messages: [assistantText("resumed"), usage(20, 40)] }, + ], + "llm1", + ); + const engine = new ContextEngine({ + llm: llm1, + environment: fakeEnvironment, + compaction: settings(), + createLLM: () => new ScriptedLLM([], "llm2"), + reconnectBackoffMs: 1, + }); + await collect(engine.run([userText("start")], { approve: allowAll })); + const aborted = new AbortController(); + aborted.abort(); + await collect( + engine.run([userText("pending question")], { signal: aborted.signal, approve: allowAll }), + ); + + const out = await collect(engine.compact()); + expect(compactionEvents(out)[1]).toMatchObject({ type: "compaction_end", status: "failed" }); + expect(llm1.calls).toHaveLength(6); + // Attempt 1 folded the carry-over in (and committed it); later attempts resend the Prompt + // alone — the absorbed carry-over never again. + expect(llm1.calls[1]!.map(textOf)).toEqual(["pending question", "COMPACT NOW"]); + for (let attempt = 2; attempt <= 5; attempt += 1) { + expect(llm1.calls[attempt]!.map(textOf)).toEqual(["COMPACT NOW"]); + } + + // The carry-over is consumed at the seam — the next run leads with the stashed repair + // alone, continuing from committed history. + await collect(engine.run([userText("run3")], { approve: allowAll })); + const run3 = llm1.calls[6]!; + expect(payloadTypes(run3)).toEqual(["tool_call_output", "text"]); + expect((run3[0]!.payload as { tool_call_id?: string }).tool_call_id).toBe("c9"); + expect(textOf(run3[1]!)).toBe("run3"); + }); + it("no compaction capability (createLLM missing) means thresholds never fire and compact() is a no-op", async () => { const llm1 = new ScriptedLLM( [{ messages: [assistantText("big"), usage(999999, 999999)] }], diff --git a/packages/core/test/replay.test.ts b/packages/core/test/replay.test.ts index fc78c0c..3294029 100644 --- a/packages/core/test/replay.test.ts +++ b/packages/core/test/replay.test.ts @@ -566,4 +566,98 @@ describe("resumeTrace regressions (PR #39 review)", () => { expect(result.sessionTurns).toBe(1); expect(result.carryOver).toEqual([]); }); + + it("absorbed mid-task compaction (committed rejections) replays with nothing left to resend", () => { + // Resume symmetry for issue #85: a mid-task compaction whose first attempt committed has + // absorbed the turn's tool output into history; the repair answering the rejected + // attempt's tool call rode the second committed attempt. After a process exit at the + // abandonment, replay must reconstruct exactly the live engine's state: everything in + // history, nothing in carry-over — the next request continues from the committed state + // without resending any tool_result. + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "ct" }), + requestEnd("completed"), + tokenUsage(usage(150), usage(150)), + toolCallOutput({ output: "tool ran", toolCallId: "ct" }), + compactionBegin({ reason: "context", mode: "summarize", context: 150, turns: 1 }), + userText("COMPACT NOW"), + requestBegin(), + assistantText("[summary]nope[/summary]"), + toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), + requestEnd("completed"), + tokenUsage(usage(160), usage(310)), + toolCallOutput({ + output: "[tool error] the compaction request expects a summary, not tool calls", + toolCallId: "c1", + stopReason: "failed", + }), + requestBegin(), + thinkingMessage("still nothing"), + requestEnd("completed"), + tokenUsage(usage(170), usage(480)), + compactionEnd({ reason: "context", mode: "summarize", status: "failed" }), + ]); + expect(result.contextClosed).toBe(false); + expect(result.pendingSummary).toBeUndefined(); + // Both committed compaction rounds enter history: the turn output was absorbed by the + // first, the repair by the second. + expect(result.history.map((m) => (m.payload as { type?: string }).type)).toEqual([ + "text", + "tool_call", + "tool_call_output", + "text", + "text", + "tool_call", + "tool_call_output", + "thinking", + ]); + // Nothing left to resend — matching the live engine, which also holds no carry-over here. + expect(result.carryOver).toEqual([]); + expect(result.sessionTurns).toBe(1); + }); + + it("absorbed mid-task compaction abandoned with an unanswered repair replays it as carry-over", () => { + // The abort-in-window shape: the rejected attempt committed (absorbing the turn output), + // its repair was written, and the process exited before any request carried the repair. + // Replay's eligibility filter keeps exactly the repair (paired with the committed, + // unanswered c1) as carry-over — the same stash the live engine holds — and does not + // reclaim the absorbed turn output. + const result = resumeTrace([ + meta(), + userText("go"), + requestBegin(), + toolCall({ name: "exec_command", arguments: "{}", toolCallId: "ct" }), + requestEnd("completed"), + tokenUsage(usage(150), usage(150)), + toolCallOutput({ output: "tool ran", toolCallId: "ct" }), + compactionBegin({ reason: "context", mode: "summarize", context: 150, turns: 1 }), + userText("COMPACT NOW"), + requestBegin(), + toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), + requestEnd("completed"), + tokenUsage(usage(160), usage(310)), + toolCallOutput({ + output: "[tool error] the compaction request expects a summary, not tool calls", + toolCallId: "c1", + stopReason: "failed", + }), + compactionEnd({ reason: "context", mode: "summarize", status: "aborted" }), + ]); + expect(result.contextClosed).toBe(false); + // The absorbed turn output sits in history (attempt 1's committed input), not carry-over. + expect(result.history.map((m) => (m.payload as { type?: string }).type)).toEqual([ + "text", + "tool_call", + "tool_call_output", + "text", + "tool_call", + ]); + expect( + result.carryOver.map((m) => (m.payload as { tool_call_id?: string }).tool_call_id), + ).toEqual(["c1"]); + expect((result.carryOver[0]!.payload as { output?: string }).output).toContain("[tool error]"); + }); }); diff --git a/packages/docs/content/agent-loop.en.md b/packages/docs/content/agent-loop.en.md index 7a6631e..36ddfb7 100644 --- a/packages/docs/content/agent-loop.en.md +++ b/packages/docs/content/agent-loop.en.md @@ -116,7 +116,7 @@ Three triggers (`compaction_begin.reason`): Two modes: `summarize` (default) appends the compaction Prompt to the old context, extracts the `[summary]`, wraps it as a `[context_summary]` user text and continues in a **fresh model context**; `discard` simply drops the old context. System markers are written as `[tag]…[/tag]`; the earlier angle-bracket form (``, ``, …) is still recognized when reading old Traces and old persisted compaction prompts. Compaction rotates the [Trace file](/sessions-and-traces) (`_002`, `_003`, …) — one Trace file always equals one complete model context. `compactability()` probes feasibility before `session.compact()` (`ok | unsupported | empty | just_compacted`). -The compaction request keeps the session's toolset **unchanged** — the request prefix (tool list included) stays byte-identical to ordinary turns, so the provider's prompt cache remains valid at the moment the context is largest. Compaction still succeeds only with a valid summary: a response that calls a tool or whose extracted summary is empty is rejected — any tool calls are answered with synthesized failed outputs (keeping `tool_use`/`tool_result` pairing intact) and the repaired request is resent immediately (a rejection is model behavior, not a transport failure, so no backoff applies), up to 5 rejected attempts; then the compaction ends `failed`, keeping the original context and Trace file until the next trigger. Transport `timeout`/`malformed` attempts follow the compaction-specific reconnect cap and backoff ladder described under "Automatic reconnect" above. +The compaction request keeps the session's toolset **unchanged** — the request prefix (tool list included) stays byte-identical to ordinary turns, so the provider's prompt cache remains valid at the moment the context is largest. Compaction still succeeds only with a valid summary: a response that calls a tool or whose extracted summary is empty is rejected — any tool calls are answered with synthesized failed outputs (keeping `tool_use`/`tool_result` pairing intact) and the repaired request is resent immediately (a rejection is model behavior, not a transport failure, so no backoff applies), up to 5 rejected attempts; then the compaction ends `failed`, keeping the original context and Trace file until the next trigger. Transport `timeout`/`malformed` attempts follow the compaction-specific reconnect cap and backoff ladder described under "Automatic reconnect" above. The first **committed** attempt — adopted or rejected — also absorbs whatever turn input was folded into the compaction request (mid-task tool results, or the carry-over a manual `/compact` folds in) into the old context's history: retries resend only the repairs and the Prompt, and the continuation after an abandoned compaction never resends the absorbed input. ## Concurrency model diff --git a/packages/docs/content/agent-loop.zh.md b/packages/docs/content/agent-loop.zh.md index d2f037d..72da422 100644 --- a/packages/docs/content/agent-loop.zh.md +++ b/packages/docs/content/agent-loop.zh.md @@ -113,7 +113,7 @@ interface CompactionSettings { 两种模式:`summarize`(默认)向旧上下文追加压缩 Prompt,提取 `[summary]` 后包装为 `[context_summary]` 用户文本,在**全新的模型上下文**中继续;`discard` 直接丢弃旧上下文。系统标记统一写作 `[tag]…[/tag]`;读取旧 Trace 与旧压缩 Prompt 时仍识别早期的尖括号形式(``、`` 等)。压缩时 [Trace 文件随之轮转](/sessions-and-traces)(`_002`、`_003`……),一个 Trace 文件恒等于一个完整模型上下文。`session.compact()` 前可用 `compactability()` 探询可行性(`ok | unsupported | empty | just_compacted`)。 -压缩请求**保持会话工具集不变**——请求前缀(含工具列表)与普通轮次逐字节一致,确保上下文最大的时刻提供商的提示词缓存依然有效。只有得到有效摘要,压缩才算成功:若响应中出现工具调用、或提取出的摘要为空,则判为无效并重试——工具调用会先以合成的失败输出逐一应答(保持 `tool_use`/`tool_result` 配对完整),修复后的请求**立即重发**(无效摘要是模型行为而非传输故障,不做退避),最多允许 5 次无效尝试,之后压缩以 `failed` 结束,保留原上下文与 Trace 文件,等待下次触发。传输层 `timeout`/`malformed` 仍走上文「自动重连」一节所述的压缩专用重连上限与退避阶梯。 +压缩请求**保持会话工具集不变**——请求前缀(含工具列表)与普通轮次逐字节一致,确保上下文最大的时刻提供商的提示词缓存依然有效。只有得到有效摘要,压缩才算成功:若响应中出现工具调用、或提取出的摘要为空,则判为无效并重试——工具调用会先以合成的失败输出逐一应答(保持 `tool_use`/`tool_result` 配对完整),修复后的请求**立即重发**(无效摘要是模型行为而非传输故障,不做退避),最多允许 5 次无效尝试,之后压缩以 `failed` 结束,保留原上下文与 Trace 文件,等待下次触发。传输层 `timeout`/`malformed` 仍走上文「自动重连」一节所述的压缩专用重连上限与退避阶梯。首个**已提交**的尝试(无论被采纳还是被判无效)同时会把折叠进压缩请求的本轮输入(任务中途的工具结果、手动 `/compact` 折叠的补发内容)吸收进旧上下文:重试只重发修复输出与压缩 Prompt,压缩放弃后的续跑也不再重发已吸收的输入。 ## 并发模型