feat(core): recover truncated tool output via the Session scratchpad (#145)
Co-authored-by: Yaowei Zheng <hiyouga@buaa.edu.cn> Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -23,7 +23,7 @@ import {
|
||||
projectDir,
|
||||
goalFilePath,
|
||||
resolveModelRef,
|
||||
scratchpadDir,
|
||||
sessionScratchpadDir,
|
||||
systemConfigPath,
|
||||
tracesDir,
|
||||
type AgentState,
|
||||
@@ -256,6 +256,7 @@ export class Agent {
|
||||
);
|
||||
|
||||
const rt = await this.buildRuntime({
|
||||
sessionId,
|
||||
workspaceDir,
|
||||
modelEntry,
|
||||
apiKey,
|
||||
@@ -292,8 +293,10 @@ export class Agent {
|
||||
createBareLLM: rt.createBareLLM,
|
||||
compaction: rt.compaction,
|
||||
// Where an input image lands when it becomes a path line (see SessionConfig.imagesDir).
|
||||
imagesDir: path.join(
|
||||
scratchpadDir(this.state.root, this.state.projectId, this.state.agentId),
|
||||
imagesDir: sessionScratchpadDir(
|
||||
this.state.root,
|
||||
this.state.projectId,
|
||||
this.state.agentId,
|
||||
sessionId,
|
||||
),
|
||||
modelHasVision: modelEntry.vision !== false,
|
||||
@@ -388,6 +391,7 @@ export class Agent {
|
||||
const thinkingLevel = this.state.systemConfig.model?.thinking_level;
|
||||
|
||||
const rt = await this.buildRuntime({
|
||||
sessionId,
|
||||
workspaceDir,
|
||||
modelEntry,
|
||||
apiKey,
|
||||
@@ -449,8 +453,10 @@ export class Agent {
|
||||
createBareLLM: rt.createBareLLM,
|
||||
compaction: rt.compaction,
|
||||
// Where an input image lands when it becomes a path line (see SessionConfig.imagesDir).
|
||||
imagesDir: path.join(
|
||||
scratchpadDir(this.state.root, this.state.projectId, this.state.agentId),
|
||||
imagesDir: sessionScratchpadDir(
|
||||
this.state.root,
|
||||
this.state.projectId,
|
||||
this.state.agentId,
|
||||
sessionId,
|
||||
),
|
||||
modelHasVision: modelEntry.vision !== false,
|
||||
@@ -492,6 +498,7 @@ export class Agent {
|
||||
* and its post-compaction rebuild factory, and the compaction config.
|
||||
*/
|
||||
private async buildRuntime(args: {
|
||||
sessionId: string;
|
||||
workspaceDir: string;
|
||||
/** This Session's Model entry: the caller (createSession / resumeSession) has already validated it exists in the config. */
|
||||
modelEntry: ModelEntry;
|
||||
@@ -511,6 +518,7 @@ export class Agent {
|
||||
compaction: CompactionSettings;
|
||||
}> {
|
||||
const {
|
||||
sessionId,
|
||||
workspaceDir,
|
||||
modelEntry,
|
||||
apiKey,
|
||||
@@ -685,6 +693,14 @@ export class Agent {
|
||||
const environment = new Environment({
|
||||
workspaceDir,
|
||||
toolConfig,
|
||||
// The Session's generic scratchpad root; Environment derives its truncated-tool-output
|
||||
// recovery directory from it.
|
||||
sessionScratchpadDir: sessionScratchpadDir(
|
||||
this.state.root,
|
||||
this.state.projectId,
|
||||
this.state.agentId,
|
||||
sessionId,
|
||||
),
|
||||
services: { subagentRunner, ...(visionDescriber ? { visionDescriber } : {}) },
|
||||
...(Object.keys(vault).length > 0 ? { vault } : {}),
|
||||
});
|
||||
|
||||
@@ -23,6 +23,7 @@
|
||||
* never empty under any circumstance**.
|
||||
* Docs: /docs/tools § "Execution contract".
|
||||
*/
|
||||
import path from "node:path";
|
||||
import { partialToolCallOutput, toolCallOutput } from "../omnimessage/index.js";
|
||||
import type { OmniMessage, StopReason } from "../omnimessage/index.js";
|
||||
import type {
|
||||
@@ -37,6 +38,13 @@ import type { BuiltinTool, ToolResult } from "./tools/types.js";
|
||||
import { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js";
|
||||
import { CommandSessionManager } from "./tools/command/index.js";
|
||||
import { SubagentSessionManager } from "./tools/subagent/index.js";
|
||||
import {
|
||||
TRUNCATED_TOOL_OUTPUT_FILE_LIMIT_BYTES,
|
||||
TruncatedToolOutputArchive,
|
||||
type TruncatedToolOutputArchiveSaveResult,
|
||||
type TruncatedToolOutputCapture,
|
||||
} from "./truncated-tool-output-archive.js";
|
||||
import { modelVisiblePath } from "../internal/model-visible-path.js";
|
||||
|
||||
/** Default cap on tool output truncation (characters). */
|
||||
const DEFAULT_MAX_OUTPUT_LENGTH = 16000;
|
||||
@@ -78,6 +86,11 @@ function noteSuffix(base: string, note: string): string {
|
||||
export class Environment implements EnvironmentInterface {
|
||||
private readonly workspaceDir: string;
|
||||
private readonly toolConfig: ToolConfig;
|
||||
/**
|
||||
* Truncated-output recovery, derived from the generic `sessionScratchpadDir` config; null for
|
||||
* standalone embedders without a Session directory (legacy truncation-only behavior).
|
||||
*/
|
||||
private readonly truncatedToolOutputArchive: TruncatedToolOutputArchive | null;
|
||||
/** Assembled built-in tools: tool name -> BuiltinTool. Only tools supported by the registry and present in config. */
|
||||
private readonly tools: Map<string, BuiltinTool>;
|
||||
/** Long-running command session registry: constructed within this Environment and shared between exec_command / input_command. */
|
||||
@@ -88,6 +101,11 @@ export class Environment implements EnvironmentInterface {
|
||||
constructor(config: EnvironmentConfig) {
|
||||
this.workspaceDir = config.workspaceDir;
|
||||
this.toolConfig = config.toolConfig;
|
||||
this.truncatedToolOutputArchive = config.sessionScratchpadDir
|
||||
? new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(config.sessionScratchpadDir, "truncated-tool-output"),
|
||||
})
|
||||
: null;
|
||||
this.tools = new Map();
|
||||
// The background session registry is created alongside Environment (one per Session) and
|
||||
// injected into whichever tools need it; all sessions are finalized together on dispose.
|
||||
@@ -216,6 +234,10 @@ export class Environment implements EnvironmentInterface {
|
||||
let selfNote: string | null = null; // Tool's self-reported end marker (e.g. exit code), appended outside truncation
|
||||
let selfImages: string[] | undefined; // Tool's self-reported images (data URL), carried via a single streamed delta and the full message
|
||||
let thrown: unknown = null;
|
||||
// Created lazily on the first over-limit text delta when this Environment has a Session
|
||||
// scratchpad. It captures the tool's complete text before Environment drops the overflow,
|
||||
// but does not alter the model/frontend stream.
|
||||
let archiveCapture: TruncatedToolOutputCapture | null = null;
|
||||
const gen = tool.execute(args, {
|
||||
workspaceDir: this.workspaceDir,
|
||||
toolCallId,
|
||||
@@ -249,6 +271,17 @@ export class Environment implements EnvironmentInterface {
|
||||
// Only takes delta content; start/stop are ignored (framing is uniformly handled by Environment).
|
||||
if (p.event_type !== "delta" || !p.output) continue;
|
||||
contentLen += p.output.length;
|
||||
const exceedsVisibleLimit = maxOutputLength > 0 && contentLen > maxOutputLength;
|
||||
if (exceedsVisibleLimit && this.truncatedToolOutputArchive) {
|
||||
if (!archiveCapture) {
|
||||
archiveCapture = this.truncatedToolOutputArchive.startCapture();
|
||||
// `streamed` is the exact prefix already accepted before this delta. Appending it
|
||||
// once, then every complete current/future delta, reconstructs the pre-truncation
|
||||
// tool text without changing what is forwarded.
|
||||
archiveCapture.append(streamed);
|
||||
}
|
||||
archiveCapture.append(p.output);
|
||||
}
|
||||
// maxOutputLength <= 0 means truncation is disabled (same semantics as timeoutMs).
|
||||
const room =
|
||||
maxOutputLength > 0 ? maxOutputLength - streamed.length : Number.POSITIVE_INFINITY;
|
||||
@@ -265,6 +298,18 @@ export class Environment implements EnvironmentInterface {
|
||||
} else if (p.type === "tool_call_output") {
|
||||
// Fallback: if the tool still produces a full message, use it as the basis for content and stop reason (not needed under the new contract).
|
||||
toolOutput = p.output ?? "";
|
||||
if (
|
||||
maxOutputLength > 0 &&
|
||||
toolOutput.length > maxOutputLength &&
|
||||
this.truncatedToolOutputArchive
|
||||
) {
|
||||
if (!archiveCapture) {
|
||||
archiveCapture = this.truncatedToolOutputArchive.startCapture();
|
||||
}
|
||||
// A compatibility tool's complete message is Environment's content basis, so it
|
||||
// also becomes the recovery basis instead of any deltas it happened to emit.
|
||||
archiveCapture.replace(toolOutput);
|
||||
}
|
||||
if (selfReported === undefined && p.stop_reason) {
|
||||
selfReported = p.stop_reason as StopReason;
|
||||
}
|
||||
@@ -293,16 +338,40 @@ export class Environment implements EnvironmentInterface {
|
||||
? contentBase.slice(0, maxOutputLength)
|
||||
: contentBase;
|
||||
const truncated = capped.length < contentBase.length || contentLen > streamed.length;
|
||||
|
||||
// Freeze the tool's terminal facts before auxiliary archive I/O. A user abort arriving
|
||||
// while the file is being written must not reclassify an already-finished tool.
|
||||
const aborted =
|
||||
signal?.aborted === true ||
|
||||
(!timedOut &&
|
||||
(selfReported === "aborted" ||
|
||||
(thrown as { name?: string } | null)?.name === "AbortError"));
|
||||
let archiveResult: TruncatedToolOutputArchiveSaveResult | null = null;
|
||||
if (truncated && archiveCapture) {
|
||||
// Both truncation paths initialize this capture at the exact point they first exceed the
|
||||
// visible cap, so a truncated call with a Session scratchpad always has one to save. A
|
||||
// standalone Environment has no capture and retains truncation-only behavior.
|
||||
archiveResult = await archiveCapture.save(name, toolCallId);
|
||||
} else {
|
||||
archiveCapture?.cancel();
|
||||
}
|
||||
|
||||
let stopReason: StopReason;
|
||||
const notes: string[] = [];
|
||||
if (truncated) {
|
||||
notes.push(`[output truncated: exceeded ${maxOutputLength} chars]`);
|
||||
if (archiveResult?.status === "saved") {
|
||||
const archivePath = modelVisiblePath(archiveResult.path);
|
||||
if (archiveResult.archiveTruncated) {
|
||||
const limitMiB = Math.ceil(TRUNCATED_TOOL_OUTPUT_FILE_LIMIT_BYTES / (1024 * 1024));
|
||||
notes.push(
|
||||
`[output archived (${limitMiB} MiB limit; head and tail kept): ${archivePath}]`,
|
||||
);
|
||||
} else {
|
||||
notes.push(`[output archived: ${archivePath}]`);
|
||||
}
|
||||
} else if (archiveResult?.status === "failed") {
|
||||
notes.push(`[output archive failed: ${archiveResult.code}]`);
|
||||
}
|
||||
}
|
||||
// The tool's self-reported end marker (e.g. exit code): appended outside the truncation —
|
||||
// if treated as a content delta it would get cut off once long output hits the cap, and the
|
||||
|
||||
@@ -27,6 +27,7 @@
|
||||
* only reports `aborted` — the interruption note is appended by Environment.
|
||||
* Docs: /docs/tools § "File tools".
|
||||
*/
|
||||
import { modelVisiblePath } from "../../internal/model-visible-path.js";
|
||||
import path from "node:path";
|
||||
import { open, realpath, stat } from "node:fs/promises";
|
||||
import { partialToolCallOutput } from "../../omnimessage/index.js";
|
||||
@@ -300,7 +301,7 @@ export function createReadFileTool(definition: ToolDefinitionConfig): BuiltinToo
|
||||
const code = (err as NodeJS.ErrnoException).code;
|
||||
if (code === "ENOENT") {
|
||||
yield delta(
|
||||
`File not found: "${filePath}". Check the path — relative paths resolve against the workspace (${ctx.workspaceDir}).`,
|
||||
`File not found: "${filePath}". Check the path — relative paths resolve against the workspace (${modelVisiblePath(ctx.workspaceDir)}).`,
|
||||
);
|
||||
} else {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
|
||||
@@ -0,0 +1,370 @@
|
||||
/**
|
||||
* TruncatedToolOutputArchive — bounded, Session-scoped recovery for text that Environment cannot
|
||||
* place in the model-visible tool result because of maxOutputLength.
|
||||
*
|
||||
* The archive is deliberately not a second tool protocol, and this module is internal:
|
||||
* Environment constructs it from the generic `EnvironmentConfig.sessionScratchpadDir` rather
|
||||
* than taking a manager object through the public config surface. Environment returns the file
|
||||
* path in the same truncated tool result seen by the frontend and the model; the model can then
|
||||
* use the existing file tools to inspect it. Files live in the Session scratchpad and are removed
|
||||
* by the host's existing Session-deletion path together with the rest of that scratchpad.
|
||||
*
|
||||
* Files are written only after a call actually exceeds maxOutputLength. A capture retains at
|
||||
* most one file's budget while the tool is streaming, then writes one UTF-8 .log file with mode
|
||||
* 0600. Small archives are exact. If a single call exceeds the per-file budget, the file keeps
|
||||
* bounded head/tail windows with an explicit gap marker.
|
||||
*/
|
||||
import { createHash } from "node:crypto";
|
||||
import { mkdir, writeFile } from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { READ_FILE_SCAN_CAP_BYTES } from "./tools/read-file.js";
|
||||
|
||||
/**
|
||||
* Maximum stored bytes for one truncated tool call. One byte of headroom below read_file's
|
||||
* 8 MiB scan cap lets that tool perform its final zero-byte read and confirm EOF.
|
||||
*/
|
||||
export const TRUNCATED_TOOL_OUTPUT_FILE_LIMIT_BYTES = READ_FILE_SCAN_CAP_BYTES - 1;
|
||||
|
||||
const ARCHIVE_GAP_MARKER = "\n[archive middle truncated]\n";
|
||||
const ARCHIVE_GAP_MARKER_BYTES = Buffer.byteLength(ARCHIVE_GAP_MARKER);
|
||||
|
||||
export type TruncatedToolOutputArchiveSaveResult =
|
||||
| {
|
||||
status: "saved";
|
||||
path: string;
|
||||
archiveTruncated: boolean;
|
||||
}
|
||||
| { status: "failed"; code: string };
|
||||
|
||||
interface TruncatedToolOutputArchiveOptions {
|
||||
rootDir: string;
|
||||
/** Test-only override; production and public SDK composition use the fixed default. */
|
||||
fileLimitBytes?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Copies one byte range into a dedicated Buffer. Using Buffer.subarray directly would retain
|
||||
* the source's entire backing ArrayBuffer, defeating the capture's memory bound.
|
||||
*/
|
||||
function copyBufferRange(buffer: Buffer, start: number, end: number): Buffer {
|
||||
const result = Buffer.alloc(Math.max(0, end - start));
|
||||
buffer.copy(result, 0, start, end);
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Copies a UTF-8-safe Buffer prefix. Moving a cut inside a multi-byte code point back to its
|
||||
* leading byte excludes that partial character rather than writing U+FFFD.
|
||||
*/
|
||||
function utf8BufferPrefix(buffer: Buffer, maxBytes: number): Buffer {
|
||||
if (maxBytes <= 0 || buffer.length === 0) return Buffer.alloc(0);
|
||||
if (buffer.length <= maxBytes) return copyBufferRange(buffer, 0, buffer.length);
|
||||
let cut = maxBytes;
|
||||
while (cut > 0 && (buffer[cut]! & 0xc0) === 0x80) cut -= 1;
|
||||
return copyBufferRange(buffer, 0, cut);
|
||||
}
|
||||
|
||||
/** Copies a UTF-8-safe Buffer suffix whose encoded size does not exceed maxBytes. */
|
||||
function utf8BufferSuffix(buffer: Buffer, maxBytes: number): Buffer {
|
||||
if (maxBytes <= 0 || buffer.length === 0) return Buffer.alloc(0);
|
||||
if (buffer.length <= maxBytes) return copyBufferRange(buffer, 0, buffer.length);
|
||||
let start = buffer.length - maxBytes;
|
||||
while (start < buffer.length && (buffer[start]! & 0xc0) === 0x80) start += 1;
|
||||
return copyBufferRange(buffer, start, buffer.length);
|
||||
}
|
||||
|
||||
/**
|
||||
* Encodes only a bounded string prefix before applying the byte cap. One UTF-16 code unit
|
||||
* contributes at least one UTF-8 byte, so maxBytes (+ one paired surrogate) is sufficient to
|
||||
* find the complete prefix without ever encoding an unbounded input delta.
|
||||
*/
|
||||
function utf8Prefix(text: string, maxBytes: number): Buffer {
|
||||
if (maxBytes <= 0 || text.length === 0) return Buffer.alloc(0);
|
||||
let end = Math.min(text.length, maxBytes);
|
||||
if (
|
||||
end < text.length &&
|
||||
end > 0 &&
|
||||
text.charCodeAt(end - 1) >= 0xd800 &&
|
||||
text.charCodeAt(end - 1) <= 0xdbff &&
|
||||
text.charCodeAt(end) >= 0xdc00 &&
|
||||
text.charCodeAt(end) <= 0xdfff
|
||||
) {
|
||||
end += 1;
|
||||
}
|
||||
return utf8BufferPrefix(Buffer.from(text.slice(0, end), "utf8"), maxBytes);
|
||||
}
|
||||
|
||||
/** Encodes only a bounded string suffix, preserving a surrogate pair at the slice boundary. */
|
||||
function utf8Suffix(text: string, maxBytes: number): Buffer {
|
||||
if (maxBytes <= 0 || text.length === 0) return Buffer.alloc(0);
|
||||
let start = Math.max(0, text.length - maxBytes);
|
||||
if (
|
||||
start > 0 &&
|
||||
text.charCodeAt(start) >= 0xdc00 &&
|
||||
text.charCodeAt(start) <= 0xdfff &&
|
||||
text.charCodeAt(start - 1) >= 0xd800 &&
|
||||
text.charCodeAt(start - 1) <= 0xdbff
|
||||
) {
|
||||
start -= 1;
|
||||
}
|
||||
return utf8BufferSuffix(Buffer.from(text.slice(start), "utf8"), maxBytes);
|
||||
}
|
||||
|
||||
/**
|
||||
* One bounded capture. Each capture independently enforces the per-file memory and disk limit.
|
||||
*/
|
||||
export class TruncatedToolOutputCapture {
|
||||
private exactChunks: Buffer[] = [];
|
||||
private exactBytes = 0;
|
||||
private head: Buffer = Buffer.alloc(0);
|
||||
/** Fixed-capacity ring storage for the rolling UTF-8 tail after promotion. */
|
||||
private tail: Buffer = Buffer.alloc(0);
|
||||
private tailStart = 0;
|
||||
private tailLength = 0;
|
||||
private archiveTruncated = false;
|
||||
private settled = false;
|
||||
/** A streamed JS string may split one UTF-16 surrogate pair across deltas. */
|
||||
private pendingHighSurrogate = "";
|
||||
|
||||
constructor(
|
||||
private readonly owner: TruncatedToolOutputArchive,
|
||||
private readonly fileLimitBytes: number,
|
||||
) {}
|
||||
|
||||
/** Appends the exact text delta produced by the tool before Environment truncates it. */
|
||||
append(text: string): void {
|
||||
if (this.settled || text.length === 0) return;
|
||||
if (this.pendingHighSurrogate) {
|
||||
const pending = this.pendingHighSurrogate;
|
||||
this.pendingHighSurrogate = "";
|
||||
const first = text.charCodeAt(0);
|
||||
if (first >= 0xdc00 && first <= 0xdfff) {
|
||||
// Join only the actual pair, not the whole new delta: concatenating a one-character
|
||||
// pending surrogate with an arbitrarily large delta would create an unbounded copy.
|
||||
this.appendStable(pending + text.slice(0, 1));
|
||||
text = text.slice(1);
|
||||
if (text.length === 0) return;
|
||||
} else {
|
||||
// The pending high surrogate is now known to be lone. Keep the current delta untouched:
|
||||
// its own final high surrogate may still pair with the following delta.
|
||||
this.appendStable(pending);
|
||||
}
|
||||
}
|
||||
const last = text.charCodeAt(text.length - 1);
|
||||
if (last >= 0xd800 && last <= 0xdbff) {
|
||||
this.pendingHighSurrogate = text.slice(-1);
|
||||
text = text.slice(0, -1);
|
||||
}
|
||||
if (text.length === 0) return;
|
||||
this.appendStable(text);
|
||||
}
|
||||
|
||||
private appendStable(text: string): void {
|
||||
const chunkBytes = Buffer.byteLength(text, "utf8");
|
||||
|
||||
if (!this.archiveTruncated) {
|
||||
if (this.exactBytes + chunkBytes <= this.fileLimitBytes) {
|
||||
// Keep this encoding inside the accepted branch. Moving Buffer.from above the size
|
||||
// guard would allocate an unbounded Buffer for one huge delta before rejecting it.
|
||||
// append() also guarantees no chunk boundary can split a still-pairable surrogate:
|
||||
// paired halves are joined first, while confirmed lone surrogates intentionally encode
|
||||
// as U+FFFD (guarded by the cross-delta Unicode test).
|
||||
this.exactChunks.push(Buffer.from(text, "utf8"));
|
||||
this.exactBytes += chunkBytes;
|
||||
return;
|
||||
}
|
||||
this.archiveTruncated = true;
|
||||
this.promoteToHeadTail(text, chunkBytes);
|
||||
return;
|
||||
}
|
||||
|
||||
this.appendTail(text, chunkBytes);
|
||||
}
|
||||
|
||||
/** Replaces the capture basis (used only by the compatibility full-message tool path). */
|
||||
replace(text: string): void {
|
||||
if (this.settled) return;
|
||||
this.exactChunks = [];
|
||||
this.exactBytes = 0;
|
||||
this.head = Buffer.alloc(0);
|
||||
this.tail = Buffer.alloc(0);
|
||||
this.tailStart = 0;
|
||||
this.tailLength = 0;
|
||||
this.archiveTruncated = false;
|
||||
this.pendingHighSurrogate = "";
|
||||
this.append(text);
|
||||
}
|
||||
|
||||
/** Writes this single-use capture to the Session archive directory. */
|
||||
async save(toolName: string, toolCallId: string): Promise<TruncatedToolOutputArchiveSaveResult> {
|
||||
if (this.settled) return { status: "failed", code: "ALREADY_SAVED" };
|
||||
if (this.pendingHighSurrogate) {
|
||||
const pending = this.pendingHighSurrogate;
|
||||
this.pendingHighSurrogate = "";
|
||||
// A truly lone high surrogate has no direct UTF-8 representation; Node's UTF-8 encoder
|
||||
// serializes it as U+FFFD, which is the same behavior writeFile(text, "utf8") would use.
|
||||
this.appendStable(pending);
|
||||
}
|
||||
this.settled = true;
|
||||
const data = this.serialized();
|
||||
return this.owner.commit(toolName, toolCallId, data, this.archiveTruncated);
|
||||
}
|
||||
|
||||
/** Discards an unfinished in-memory capture without writing a file. */
|
||||
cancel(): void {
|
||||
if (this.settled) return;
|
||||
this.settled = true;
|
||||
}
|
||||
|
||||
private promoteToHeadTail(text: string, chunkBytes: number): void {
|
||||
const contentBudget = Math.max(0, this.fileLimitBytes - ARCHIVE_GAP_MARKER_BYTES);
|
||||
const headBudget = Math.floor(contentBudget / 2);
|
||||
const tailBudget = contentBudget - headBudget;
|
||||
const exact = Buffer.concat(this.exactChunks);
|
||||
|
||||
this.head =
|
||||
this.exactBytes >= headBudget
|
||||
? utf8BufferPrefix(exact, headBudget)
|
||||
: Buffer.concat([exact, utf8Prefix(text, headBudget - this.exactBytes)]);
|
||||
const initialTail =
|
||||
chunkBytes >= tailBudget
|
||||
? utf8Suffix(text, tailBudget)
|
||||
: Buffer.concat([
|
||||
utf8BufferSuffix(exact, tailBudget - chunkBytes),
|
||||
Buffer.from(text, "utf8"),
|
||||
]);
|
||||
this.resetTail(initialTail, tailBudget);
|
||||
this.exactChunks = [];
|
||||
this.exactBytes = 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Appends one stable delta to the rolling tail without rebuilding the retained window.
|
||||
* Encoding remains below the tail budget: a delta at least that large takes the bounded
|
||||
* string-suffix path instead of allocating a Buffer for the whole delta.
|
||||
*/
|
||||
private appendTail(text: string, chunkBytes: number): void {
|
||||
const tailBudget = this.tailBudget();
|
||||
if (tailBudget <= 0) {
|
||||
this.resetTail(Buffer.alloc(0), 0);
|
||||
return;
|
||||
}
|
||||
if (chunkBytes >= tailBudget) {
|
||||
this.resetTail(utf8Suffix(text, tailBudget), tailBudget);
|
||||
return;
|
||||
}
|
||||
|
||||
const chunk = Buffer.from(text, "utf8");
|
||||
const overflow = Math.max(0, this.tailLength + chunk.length - tailBudget);
|
||||
this.tailStart = (this.tailStart + overflow) % tailBudget;
|
||||
const nextLength = Math.min(tailBudget, this.tailLength + chunk.length);
|
||||
const writeStart = (this.tailStart + nextLength - chunk.length) % tailBudget;
|
||||
const firstLength = Math.min(chunk.length, tailBudget - writeStart);
|
||||
chunk.copy(this.tail, writeStart, 0, firstLength);
|
||||
if (firstLength < chunk.length) {
|
||||
chunk.copy(this.tail, 0, firstLength);
|
||||
}
|
||||
this.tailLength = nextLength;
|
||||
}
|
||||
|
||||
/** Reinitializes the rolling tail from one already-bounded, code-point-aligned suffix. */
|
||||
private resetTail(buffer: Buffer, tailBudget: number): void {
|
||||
if (tailBudget <= 0) {
|
||||
this.tail = Buffer.alloc(0);
|
||||
this.tailStart = 0;
|
||||
this.tailLength = 0;
|
||||
return;
|
||||
}
|
||||
this.tail = Buffer.allocUnsafe(tailBudget);
|
||||
buffer.copy(this.tail);
|
||||
this.tailStart = 0;
|
||||
this.tailLength = buffer.length;
|
||||
}
|
||||
|
||||
private tailBudget(): number {
|
||||
const contentBudget = Math.max(0, this.fileLimitBytes - ARCHIVE_GAP_MARKER_BYTES);
|
||||
return contentBudget - Math.floor(contentBudget / 2);
|
||||
}
|
||||
|
||||
/** Copies the logical ring suffix once, dropping a leading partial UTF-8 code point. */
|
||||
private serializedTail(): Buffer {
|
||||
if (this.tailLength === 0) return Buffer.alloc(0);
|
||||
const capacity = this.tail.length;
|
||||
let skip = 0;
|
||||
while (
|
||||
skip < this.tailLength &&
|
||||
(this.tail[(this.tailStart + skip) % capacity]! & 0xc0) === 0x80
|
||||
) {
|
||||
skip += 1;
|
||||
}
|
||||
const length = this.tailLength - skip;
|
||||
const result = Buffer.allocUnsafe(length);
|
||||
const readStart = (this.tailStart + skip) % capacity;
|
||||
const firstLength = Math.min(length, capacity - readStart);
|
||||
this.tail.copy(result, 0, readStart, readStart + firstLength);
|
||||
if (firstLength < length) {
|
||||
this.tail.copy(result, firstLength, 0, length - firstLength);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
private serialized(): Buffer {
|
||||
if (!this.archiveTruncated) return Buffer.concat(this.exactChunks);
|
||||
return Buffer.concat([
|
||||
this.head,
|
||||
Buffer.from(ARCHIVE_GAP_MARKER, "utf8"),
|
||||
this.serializedTail(),
|
||||
]);
|
||||
}
|
||||
}
|
||||
|
||||
export class TruncatedToolOutputArchive {
|
||||
private readonly rootDir: string;
|
||||
private readonly fileLimitBytes: number;
|
||||
|
||||
constructor(opts: TruncatedToolOutputArchiveOptions) {
|
||||
// The explicit gap marker is part of every bounded head/tail archive, so even internal
|
||||
// test overrides must leave enough room for it; production stays just below 8 MiB.
|
||||
this.fileLimitBytes = Math.max(
|
||||
ARCHIVE_GAP_MARKER_BYTES,
|
||||
opts.fileLimitBytes ?? TRUNCATED_TOOL_OUTPUT_FILE_LIMIT_BYTES,
|
||||
);
|
||||
this.rootDir = opts.rootDir;
|
||||
}
|
||||
|
||||
/** Starts one independently bounded capture; the directory remains lazy until save(). */
|
||||
startCapture(): TruncatedToolOutputCapture {
|
||||
return new TruncatedToolOutputCapture(this, this.fileLimitBytes);
|
||||
}
|
||||
|
||||
/** Internal commit path used by TruncatedToolOutputCapture. */
|
||||
async commit(
|
||||
toolName: string,
|
||||
toolCallId: string,
|
||||
data: Buffer,
|
||||
archiveTruncated: boolean,
|
||||
): Promise<TruncatedToolOutputArchiveSaveResult> {
|
||||
const safeToolName = toolName.replace(/[^a-zA-Z0-9_-]/g, "_").slice(0, 48) || "tool";
|
||||
const idHash = createHash("sha256").update(toolCallId).digest("hex").slice(0, 16);
|
||||
const filePath = path.join(this.rootDir, `${safeToolName}-${idHash}.log`);
|
||||
try {
|
||||
// Create shared Session ancestors with their existing/default policy, then apply the
|
||||
// archive's private directory mode only to the archive directory itself.
|
||||
await mkdir(path.dirname(this.rootDir), { recursive: true });
|
||||
await mkdir(this.rootDir, { recursive: true, mode: 0o700 });
|
||||
await writeFile(filePath, data, { flag: "wx", mode: 0o600 });
|
||||
return {
|
||||
status: "saved",
|
||||
path: filePath,
|
||||
archiveTruncated,
|
||||
};
|
||||
} catch (err) {
|
||||
const rawCode = (err as { code?: unknown }).code;
|
||||
const code = typeof rawCode === "string" ? rawCode : "UNKNOWN";
|
||||
process.stderr.write(
|
||||
`[penguin] tool "${toolName}" truncated output archive write failed (${code}).\n`,
|
||||
);
|
||||
return { status: "failed", code };
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -22,6 +22,7 @@
|
||||
* budget numbers. The embedded `objective` value is user data, which is why the closing tag
|
||||
* is matched line-anchored (see markers/goal-block.ts).
|
||||
*/
|
||||
import { modelVisiblePath } from "../internal/model-visible-path.js";
|
||||
import { markerBlock, MARKER_TAGS } from "../omnimessage/markers/index.js";
|
||||
import { serializeGoalFile, UNLIMITED_BUDGET } from "./goal-file.js";
|
||||
|
||||
@@ -43,7 +44,7 @@ export interface GoalPromptArgs {
|
||||
/** The goal-file paragraph shared by both blocks: path, the status protocol, and the file's content. */
|
||||
function goalFileLines(args: GoalPromptArgs): string[] {
|
||||
return [
|
||||
`Goal file: ${args.goalFilePath}`,
|
||||
`Goal file: ${modelVisiblePath(args.goalFilePath)}`,
|
||||
"You may modify ONLY the `status` field of this file, and only to `complete` or",
|
||||
"`blocked`; the system reads it after every round. Its content:",
|
||||
"",
|
||||
|
||||
@@ -52,6 +52,9 @@ export type { SessionTitleResult } from "./internal/session-title.js";
|
||||
// re-exported, because the server appends `[attached file: …]` lines for the composer's
|
||||
// uploads and both producers must place them identically (see the markers module).
|
||||
export { appendAttachmentLines } from "./internal/session-support.js";
|
||||
// Model-visible path spelling (forward slashes on Windows); the server uses it for its
|
||||
// [attached file: ...] lines so every path the model reads has one spelling per platform.
|
||||
export { modelVisiblePath } from "./internal/model-visible-path.js";
|
||||
export { Agent, createAgent } from "./agent.js";
|
||||
export type { CreateAgentOptions, CreateSessionOptions, ResumeSessionOptions } from "./agent.js";
|
||||
|
||||
|
||||
@@ -275,6 +275,14 @@ export interface EnvironmentServices {
|
||||
export interface EnvironmentConfig {
|
||||
workspaceDir: string;
|
||||
toolConfig: ToolConfig;
|
||||
/**
|
||||
* This Session's private scratchpad directory (`scratchpad/<sessionId>`), the generic
|
||||
* Session-scoped storage root for Environment by-products. Currently it backs
|
||||
* truncated-tool-output recovery: output beyond an entry's `maxOutputLength` is saved under
|
||||
* `<sessionScratchpadDir>/truncated-tool-output/`. Agent Sessions always pass it; standalone
|
||||
* embedders without a stable Session directory omit it and keep truncation-only behavior.
|
||||
*/
|
||||
sessionScratchpadDir?: string;
|
||||
/** Runtime services (optional); Environment forwards these to each tool factory to use as needed. */
|
||||
services?: EnvironmentServices;
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
/**
|
||||
* Model-visible spelling of an absolute path.
|
||||
*
|
||||
* Every path core composes for the model to read — the system prompt's App Data Dir and CWD
|
||||
* lines, `[attached image/file: …]` lines, the goal-file line, truncated-output recovery notes —
|
||||
* goes through this helper, because the model re-emits those spellings into JSON tool arguments
|
||||
* and shell commands. On Windows that spelling uses forward slashes: Node's fs APIs accept them,
|
||||
* `exec_command` runs through (Git) Bash, and the form has no JSON backslash-escaping ambiguity.
|
||||
* Harness-composed paths are ordinary absolute paths (never `\\?\`-prefixed), so the swap is
|
||||
* lossless. POSIX paths pass through untouched — a backslash is a valid filename character there.
|
||||
*/
|
||||
export function modelVisiblePath(filePath: string): string {
|
||||
return process.platform === "win32" ? filePath.replaceAll("\\", "/") : filePath;
|
||||
}
|
||||
@@ -11,6 +11,7 @@ import { randomBytes, randomUUID } from "node:crypto";
|
||||
import { formatLocalDate } from "./dates.js";
|
||||
import { sessionShell } from "../environment/tools/command/shell.js";
|
||||
import type { SessionEnvironmentValues } from "../state/agent-state.js";
|
||||
import { modelVisiblePath } from "./model-visible-path.js";
|
||||
import { workspacesDir } from "../state/index.js";
|
||||
import { attachedImageLine, isWholeOriginBlock, userText } from "../omnimessage/index.js";
|
||||
import type { OmniMessage } from "../omnimessage/index.js";
|
||||
@@ -41,9 +42,11 @@ export function sessionEnvironment(
|
||||
): SessionEnvironment {
|
||||
return {
|
||||
sessionId,
|
||||
cwd: workspaceDir,
|
||||
// Model-visible spelling (forward slashes on Windows): the model composes tool arguments
|
||||
// and shell commands from these two lines, so they must be safe in both contexts.
|
||||
cwd: modelVisiblePath(workspaceDir),
|
||||
agentId: ids.agentId,
|
||||
projectDir: ids.projectDir,
|
||||
projectDir: modelVisiblePath(ids.projectDir),
|
||||
provider: ids.provider,
|
||||
modelId: ids.modelId,
|
||||
platform: process.platform,
|
||||
@@ -167,7 +170,7 @@ export async function imagesToScratchpadPaths(
|
||||
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
|
||||
}
|
||||
}
|
||||
lines.push(attachedImageLine(file));
|
||||
lines.push(attachedImageLine(modelVisiblePath(file)));
|
||||
}
|
||||
|
||||
return appendAttachmentLines(
|
||||
|
||||
@@ -67,6 +67,21 @@ export function workspacesDir(root: string, projectId: string, agentId: string):
|
||||
return path.join(agentDir(root, projectId, agentId), "workspaces");
|
||||
}
|
||||
|
||||
/**
|
||||
* `<agentDir>/scratchpad/<sessionId>`, one Session's private scratchpad directory. The single
|
||||
* Session-scoped storage root shared by every by-product bound to that Session: input images
|
||||
* saved as path lines, the goal-mode control file, and Environment's truncated-tool-output
|
||||
* recovery files. Deleted together with the Session by the existing scratchpad cleanup path.
|
||||
*/
|
||||
export function sessionScratchpadDir(
|
||||
root: string,
|
||||
projectId: string,
|
||||
agentId: string,
|
||||
sessionId: string,
|
||||
): string {
|
||||
return path.join(scratchpadDir(root, projectId, agentId), sessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* `<agentDir>/scratchpad/<sessionId>/GOAL.yaml`, the goal-mode control file of one Session
|
||||
* (sibling of the model's PLAN.md convention; see goal/goal-file.ts for field ownership).
|
||||
@@ -77,7 +92,7 @@ export function goalFilePath(
|
||||
agentId: string,
|
||||
sessionId: string,
|
||||
): string {
|
||||
return path.join(scratchpadDir(root, projectId, agentId), sessionId, "GOAL.yaml");
|
||||
return path.join(sessionScratchpadDir(root, projectId, agentId, sessionId), "GOAL.yaml");
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -13,6 +13,7 @@ import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import { createAgent } from "../src/index.js";
|
||||
import { formatSessionId } from "../src/internal/session-support.js";
|
||||
import { projectDir } from "../src/state/paths.js";
|
||||
import { modelVisiblePath } from "../src/internal/model-visible-path.js";
|
||||
import { stubProviderKeys } from "./provider-keys.js";
|
||||
|
||||
let tmpRoot: string;
|
||||
@@ -99,7 +100,11 @@ describe("Agent.createSession session id + no .penguin symlink", () => {
|
||||
const session = await agent.createSession({ workspaceDir: ws });
|
||||
const prompt = (session.metaMessage.payload as { system_prompt: string }).system_prompt;
|
||||
expect(prompt).toContain(`Agent ID: ${agent.state.agentId}`);
|
||||
expect(prompt).toContain(`App Data Dir: ${projectDir(tmpRoot, agent.state.projectId)}`);
|
||||
// The prompt shows the model-visible spelling (forward slashes on Windows), not path.join's.
|
||||
expect(prompt).toContain(
|
||||
`App Data Dir: ${modelVisiblePath(projectDir(tmpRoot, agent.state.projectId))}`,
|
||||
);
|
||||
expect(prompt).toContain(`CWD: ${modelVisiblePath(ws)}`);
|
||||
expect(prompt).not.toContain(".penguin");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -152,6 +152,11 @@ async function readFileEventually(
|
||||
}
|
||||
}
|
||||
|
||||
/** Extracts the plain archive path from the note (the path is always last before `]`). */
|
||||
function recoveryPath(output: string): string | undefined {
|
||||
return output.match(/\[output archived[^:]*: ([^\]]+)\]/)?.[1];
|
||||
}
|
||||
|
||||
describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => {
|
||||
let workspace: string;
|
||||
let traces: string;
|
||||
@@ -210,6 +215,112 @@ describe("ContextEngine ReAct loop (mock LLM, approve callback)", () => {
|
||||
expect(recordedTypes.some((t) => t?.startsWith("partial_"))).toBe(false);
|
||||
});
|
||||
|
||||
it("lets the next Agent turn recover truncated text, keeps UI == Agent output, and preserves it after Task end", async () => {
|
||||
const NAME = "__recoverable_text_tool__";
|
||||
const source = `BEGIN\n${"detail\n".repeat(100)}FINAL ANSWER\n`;
|
||||
BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({
|
||||
name: NAME,
|
||||
definition,
|
||||
async *execute(_args, ctx) {
|
||||
yield partialToolCallOutput({
|
||||
eventType: "delta",
|
||||
output: source.slice(0, 200),
|
||||
toolCallId: ctx.toolCallId,
|
||||
});
|
||||
yield partialToolCallOutput({
|
||||
eventType: "delta",
|
||||
output: source.slice(200),
|
||||
toolCallId: ctx.toolCallId,
|
||||
});
|
||||
},
|
||||
});
|
||||
try {
|
||||
let calls = 0;
|
||||
let agentVisibleOutput = "";
|
||||
let recoveredPath = "";
|
||||
let recoveredDuringTask = "";
|
||||
const llm: LLMInterface = {
|
||||
async *streamGenerate(params): AsyncGenerator<OmniMessage, LLMOutcome> {
|
||||
calls += 1;
|
||||
if (calls === 1) {
|
||||
yield toolCall({
|
||||
name: NAME,
|
||||
arguments: "{}",
|
||||
toolCallId: "recover-call",
|
||||
stopReason: "completed",
|
||||
});
|
||||
yield tokenUsage(emptyTokenCounts(), {
|
||||
cache_read: 0,
|
||||
cache_write: 0,
|
||||
output: 1,
|
||||
total: 1,
|
||||
});
|
||||
return { status: "completed" };
|
||||
}
|
||||
const toolResult = params.newMessages.find(
|
||||
(m) => (m.payload as { type?: string }).type === "tool_call_output",
|
||||
);
|
||||
agentVisibleOutput = (toolResult?.payload as { output?: string }).output ?? "";
|
||||
recoveredPath = recoveryPath(agentVisibleOutput) ?? "";
|
||||
recoveredDuringTask = await readFile(recoveredPath, "utf8");
|
||||
yield assistantText("Recovered the final answer.");
|
||||
yield tokenUsage(emptyTokenCounts(), {
|
||||
cache_read: 0,
|
||||
cache_write: 0,
|
||||
output: 1,
|
||||
total: 2,
|
||||
});
|
||||
return { status: "completed" };
|
||||
},
|
||||
};
|
||||
const environment = new Environment({
|
||||
workspaceDir: workspace,
|
||||
toolConfig: {
|
||||
customTools: [
|
||||
{ name: NAME, description: "recover", permission: "r", maxOutputLength: 40 },
|
||||
],
|
||||
mcpServers: [],
|
||||
},
|
||||
sessionScratchpadDir: join(workspace, "session-scratchpad"),
|
||||
});
|
||||
const trace = new Writer({ tracesDir: traces, sessionId: "sess_truncated_output" });
|
||||
const engine = new ContextEngine({ llm, environment, trace });
|
||||
const all = await collectRun(engine, [userText("recover it")], allowAll);
|
||||
|
||||
expect(recoveredDuringTask).toBe(source);
|
||||
const frontendComplete = all.find(
|
||||
(m) =>
|
||||
(m.payload as { type?: string }).type === "tool_call_output" &&
|
||||
(m.payload as { tool_call_id?: string }).tool_call_id === "recover-call",
|
||||
);
|
||||
expect((frontendComplete!.payload as { output: string }).output).toBe(agentVisibleOutput);
|
||||
const frontendStream = all
|
||||
.filter(
|
||||
(m) =>
|
||||
(m.payload as { type?: string }).type === "partial_tool_call_output" &&
|
||||
(m.payload as { tool_call_id?: string }).tool_call_id === "recover-call" &&
|
||||
(m.payload as { event_type?: string }).event_type === "delta",
|
||||
)
|
||||
.map((m) => (m.payload as { output?: string }).output ?? "")
|
||||
.join("");
|
||||
expect(frontendStream).toBe(agentVisibleOutput);
|
||||
|
||||
const recorded = await readTrace(trace.currentPath());
|
||||
const tracedOutput = recorded.find(
|
||||
(m) =>
|
||||
(m.payload as { type?: string }).type === "tool_call_output" &&
|
||||
(m.payload as { tool_call_id?: string }).tool_call_id === "recover-call",
|
||||
);
|
||||
expect((tracedOutput!.payload as { output: string }).output).toBe(agentVisibleOutput);
|
||||
|
||||
expect(await readFile(recoveredPath, "utf8")).toBe(source);
|
||||
environment.dispose();
|
||||
expect(await readFile(recoveredPath, "utf8")).toBe(source);
|
||||
} finally {
|
||||
delete BUILTIN_TOOL_FACTORIES[NAME];
|
||||
}
|
||||
});
|
||||
|
||||
it("a slow tool delays the run only by its own latency: the loop adds no waits, timers, or dropped wakes", async () => {
|
||||
// Regression pin for the ci-windows timeout of the test above (goal-mode PR #66's
|
||||
// merge-ref run): on one cold Windows runner the suite-start burst of first Git-Bash
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import { mkdtemp, readFile, rm } from "node:fs/promises";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { access, mkdir, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import { Environment } from "../src/environment/index.js";
|
||||
import { TruncatedToolOutputCapture } from "../src/environment/truncated-tool-output-archive.js";
|
||||
import {
|
||||
partialToolCallOutput,
|
||||
toolCall,
|
||||
@@ -71,6 +72,11 @@ function payloadTypes(messages: OmniMessage[]): string[] {
|
||||
return messages.map((m) => (m.payload as { type?: string }).type ?? "");
|
||||
}
|
||||
|
||||
/** Extracts the plain archive path from the note (the path is always last before `]`). */
|
||||
function recoveryPath(output: string): string | undefined {
|
||||
return output.match(/\[output archived[^:]*: ([^\]]+)\]/)?.[1];
|
||||
}
|
||||
|
||||
let tmp: string;
|
||||
let originalHome: string | undefined;
|
||||
|
||||
@@ -251,11 +257,31 @@ describe("Environment.executeTool — edit file", () => {
|
||||
});
|
||||
|
||||
describe("Environment.executeTool — maxOutputLength truncation", () => {
|
||||
it("truncates front-to-back at the limit with a trailing marker; stream == complete", async () => {
|
||||
it("keeps standalone Environment's legacy truncation behavior without archive config", async () => {
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ maxOutputLength: 5 })),
|
||||
});
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "exec_command",
|
||||
arguments: JSON.stringify({ cmd: "printf 'abcdefghijklmnopqrstuvwxyz'" }),
|
||||
toolCallId: "call_standalone_truncation",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
const output = (messages[messages.length - 1]!.payload as { output: string }).output;
|
||||
expect(output).toContain("[output truncated: exceeded 5 chars]");
|
||||
expect(output).not.toContain("[output archived:");
|
||||
});
|
||||
|
||||
it("truncates front-to-back, archives the received output, and keeps stream == complete", async () => {
|
||||
const maxOutputLength = 50;
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ maxOutputLength })),
|
||||
sessionScratchpadDir: path.join(tmp, "scratch"),
|
||||
});
|
||||
|
||||
const messages = await collect(
|
||||
@@ -276,7 +302,10 @@ describe("Environment.executeTool — maxOutputLength truncation", () => {
|
||||
expect(output.startsWith("1\n2\n3\n")).toBe(true);
|
||||
const marker = `[output truncated: exceeded ${maxOutputLength} chars]`;
|
||||
expect(output).toContain(marker);
|
||||
expect(output.length).toBeLessThanOrEqual(maxOutputLength + marker.length + 1);
|
||||
expect(output.indexOf(marker)).toBe(maxOutputLength + 1);
|
||||
const savedPath = recoveryPath(output);
|
||||
expect(savedPath).toBeDefined();
|
||||
expect(await readFile(savedPath!, "utf8")).toMatch(/100000\r?\n$/);
|
||||
// Even when truncated, concatenating the streamed deltas == the complete content (the
|
||||
// excess part is never forwarded).
|
||||
const streamed = messages
|
||||
@@ -291,9 +320,12 @@ describe("Environment.executeTool — maxOutputLength truncation", () => {
|
||||
});
|
||||
|
||||
it("maxOutputLength <= 0 disables truncation", async () => {
|
||||
const sessionScratchpadDir = path.join(tmp, "scratch");
|
||||
const truncatedToolOutputRoot = path.join(sessionScratchpadDir, "truncated-tool-output");
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ maxOutputLength: 0 })),
|
||||
sessionScratchpadDir,
|
||||
});
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
@@ -307,6 +339,330 @@ describe("Environment.executeTool — maxOutputLength truncation", () => {
|
||||
const output = (messages[messages.length - 1]!.payload as { output: string }).output;
|
||||
expect(output).toContain("100");
|
||||
expect(output).not.toContain("[output truncated");
|
||||
await expect(access(truncatedToolOutputRoot)).rejects.toThrow();
|
||||
});
|
||||
|
||||
it("saves exact overflow for the Session without changing stream == complete", async () => {
|
||||
const maxOutputLength = 50;
|
||||
const workspaceDir = path.join(tmp, "workspace");
|
||||
await mkdir(workspaceDir);
|
||||
const sessionScratchpadDir = path.join(tmp, "session-scratchpad");
|
||||
const truncatedToolOutputRoot = path.join(sessionScratchpadDir, "truncated-tool-output");
|
||||
const toolConfig: ToolConfig = {
|
||||
customTools: [
|
||||
execTool({ maxOutputLength }),
|
||||
{
|
||||
name: "read_file",
|
||||
description: "Read a text file.",
|
||||
permission: "r",
|
||||
maxOutputLength: 64_000,
|
||||
},
|
||||
],
|
||||
mcpServers: [],
|
||||
};
|
||||
const env = new Environment({
|
||||
workspaceDir,
|
||||
toolConfig,
|
||||
sessionScratchpadDir,
|
||||
});
|
||||
const expected = `BEGIN\n${"x".repeat(200)}\nEND\n`;
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "exec_command",
|
||||
arguments: JSON.stringify({
|
||||
cmd: `node -e ${JSON.stringify(`process.stdout.write(${JSON.stringify(expected)})`)}`,
|
||||
}),
|
||||
toolCallId: "call_recoverable",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
|
||||
const complete = messages[messages.length - 1]!.payload as {
|
||||
output: string;
|
||||
stop_reason?: string;
|
||||
};
|
||||
expect(complete.stop_reason).toBe("completed");
|
||||
expect(complete.output.startsWith(expected.slice(0, maxOutputLength))).toBe(true);
|
||||
const savedPath = recoveryPath(complete.output);
|
||||
expect(savedPath).toBeDefined();
|
||||
if (process.platform === "win32") {
|
||||
expect(savedPath).not.toContain("\\");
|
||||
}
|
||||
expect(await readFile(savedPath!, "utf8")).toBe(expected);
|
||||
if (process.platform !== "win32") {
|
||||
expect((await stat(savedPath!)).mode & 0o777).toBe(0o600);
|
||||
}
|
||||
|
||||
const streamed = messages
|
||||
.filter(
|
||||
(m) =>
|
||||
(m.payload as { type?: string }).type === "partial_tool_call_output" &&
|
||||
(m.payload as { event_type?: string }).event_type === "delta",
|
||||
)
|
||||
.map((m) => (m.payload as { output?: string }).output ?? "")
|
||||
.join("");
|
||||
// Core product invariant: Web/CLI consume this stream, while the Agent receives the
|
||||
// complete result. The archive path is present identically on both sides.
|
||||
expect(streamed).toBe(complete.output);
|
||||
|
||||
// The production recovery directory is outside the Workspace. Pin that the existing
|
||||
// read_file tool accepts the absolute path, so no dedicated recovery tool is needed.
|
||||
const readMessages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "read_file",
|
||||
arguments: JSON.stringify({ file_path: savedPath }),
|
||||
toolCallId: "read_recovery",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
const readOutput = (readMessages[readMessages.length - 1]!.payload as { output: string })
|
||||
.output;
|
||||
expect(readOutput).toContain("BEGIN");
|
||||
expect(readOutput).toContain("END");
|
||||
|
||||
env.dispose();
|
||||
expect(await readFile(savedPath!, "utf8")).toBe(expected);
|
||||
await expect(access(truncatedToolOutputRoot)).resolves.toBeUndefined();
|
||||
|
||||
// Resuming the Session creates a fresh Environment over the same scratchpad. The path
|
||||
// recorded in Trace must still work with the ordinary read_file tool.
|
||||
const resumedEnv = new Environment({
|
||||
workspaceDir,
|
||||
toolConfig,
|
||||
sessionScratchpadDir,
|
||||
});
|
||||
const resumedRead = await collect(
|
||||
resumedEnv.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "read_file",
|
||||
arguments: JSON.stringify({ file_path: savedPath }),
|
||||
toolCallId: "read_after_resume",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
expect((resumedRead[resumedRead.length - 1]!.payload as { output: string }).output).toContain(
|
||||
"END",
|
||||
);
|
||||
resumedEnv.dispose();
|
||||
});
|
||||
|
||||
it("does not create a Session output directory for a result that fits the visible limit", async () => {
|
||||
const sessionScratchpadDir = path.join(tmp, "scratch");
|
||||
const truncatedToolOutputRoot = path.join(sessionScratchpadDir, "truncated-tool-output");
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ maxOutputLength: 100 })),
|
||||
sessionScratchpadDir,
|
||||
});
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "exec_command",
|
||||
arguments: JSON.stringify({ cmd: "printf short" }),
|
||||
toolCallId: "call_short",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
const output = (messages[messages.length - 1]!.payload as { output: string }).output;
|
||||
expect(output).not.toContain("[output archived:");
|
||||
await expect(access(truncatedToolOutputRoot)).rejects.toThrow();
|
||||
});
|
||||
|
||||
it("keeps the original tool outcome when the recovery file cannot be written", async () => {
|
||||
const sessionScratchpadDir = path.join(tmp, "scratchpad-is-a-file");
|
||||
await writeFile(sessionScratchpadDir, "occupied", "utf8");
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ maxOutputLength: 20 })),
|
||||
sessionScratchpadDir,
|
||||
});
|
||||
const stderr: string[] = [];
|
||||
const stderrSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
|
||||
stderr.push(String(chunk));
|
||||
return true;
|
||||
});
|
||||
let messages: OmniMessage[];
|
||||
try {
|
||||
messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "exec_command",
|
||||
arguments: JSON.stringify({ cmd: "printf 'abcdefghijklmnopqrstuvwxyz'" }),
|
||||
toolCallId: "call_archive_failure",
|
||||
}),
|
||||
}),
|
||||
);
|
||||
} finally {
|
||||
stderrSpy.mockRestore();
|
||||
}
|
||||
const complete = messages[messages.length - 1]!.payload as {
|
||||
output: string;
|
||||
stop_reason?: string;
|
||||
};
|
||||
expect(complete.output).toMatch(/\[output archive failed: [^\]]+\]/);
|
||||
expect(complete.stop_reason).toBe("completed");
|
||||
expect(stderr.join("")).toMatch(
|
||||
/\[penguin\] tool "exec_command" truncated output archive write failed \([^)]+\)\./,
|
||||
);
|
||||
const streamed = messages
|
||||
.filter(
|
||||
(m) =>
|
||||
(m.payload as { type?: string }).type === "partial_tool_call_output" &&
|
||||
(m.payload as { event_type?: string }).event_type === "delta",
|
||||
)
|
||||
.map((m) => (m.payload as { output?: string }).output ?? "")
|
||||
.join("");
|
||||
expect(streamed).toBe(complete.output);
|
||||
});
|
||||
|
||||
it("freezes the completed outcome before auxiliary archive I/O", async () => {
|
||||
const controller = new AbortController();
|
||||
const originalSave = TruncatedToolOutputCapture.prototype.save;
|
||||
const saveSpy = vi
|
||||
.spyOn(TruncatedToolOutputCapture.prototype, "save")
|
||||
.mockImplementation(function (this: TruncatedToolOutputCapture, toolName, toolCallId) {
|
||||
// The tool has already reached its terminal state when save() starts. This late abort
|
||||
// must not retroactively turn a completed tool into an aborted one.
|
||||
controller.abort();
|
||||
return originalSave.call(this, toolName, toolCallId);
|
||||
});
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: makeToolConfig(execTool({ maxOutputLength: 5 })),
|
||||
sessionScratchpadDir: path.join(tmp, "scratch"),
|
||||
});
|
||||
try {
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({
|
||||
name: "exec_command",
|
||||
arguments: JSON.stringify({ cmd: "printf 'abcdefghijklmnopqrstuvwxyz'" }),
|
||||
toolCallId: "call_late_abort",
|
||||
}),
|
||||
signal: controller.signal,
|
||||
}),
|
||||
);
|
||||
const complete = messages[messages.length - 1]!.payload as {
|
||||
output: string;
|
||||
stop_reason?: string;
|
||||
};
|
||||
expect(saveSpy).toHaveBeenCalledOnce();
|
||||
expect(controller.signal.aborted).toBe(true);
|
||||
expect(complete.stop_reason).toBe("completed");
|
||||
expect(complete.output).toContain("[output archived:");
|
||||
expect(complete.output).not.toContain("[interrupted: tool aborted by user]");
|
||||
} finally {
|
||||
saveSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
it("bounds a very large archive and preserves its UTF-8 head and tail", async () => {
|
||||
const NAME = "__large_text_tool__";
|
||||
BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({
|
||||
name: NAME,
|
||||
definition,
|
||||
async *execute(_args, ctx) {
|
||||
yield partialToolCallOutput({
|
||||
eventType: "delta",
|
||||
output: "BEGIN-企鹅\n",
|
||||
toolCallId: ctx.toolCallId,
|
||||
});
|
||||
yield partialToolCallOutput({
|
||||
eventType: "delta",
|
||||
output: "x".repeat(8 * 1024 * 1024 + 100_000),
|
||||
toolCallId: ctx.toolCallId,
|
||||
});
|
||||
yield partialToolCallOutput({
|
||||
eventType: "delta",
|
||||
output: "\n-END-🐧",
|
||||
toolCallId: ctx.toolCallId,
|
||||
});
|
||||
},
|
||||
});
|
||||
try {
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: {
|
||||
customTools: [{ name: NAME, description: "large", permission: "r", maxOutputLength: 50 }],
|
||||
mcpServers: [],
|
||||
},
|
||||
sessionScratchpadDir: path.join(tmp, "scratch"),
|
||||
});
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "call_huge" }),
|
||||
}),
|
||||
);
|
||||
const complete = messages[messages.length - 1]!.payload as { output: string };
|
||||
expect(complete.output).toContain("head and tail kept");
|
||||
const savedPath = recoveryPath(complete.output);
|
||||
expect(savedPath).toBeDefined();
|
||||
const archived = await readFile(savedPath!, "utf8");
|
||||
expect(Buffer.byteLength(archived, "utf8")).toBeLessThanOrEqual(8 * 1024 * 1024);
|
||||
expect(archived).toContain("BEGIN-企鹅");
|
||||
expect(archived).toContain("-END-🐧");
|
||||
expect(archived).toContain("[archive middle truncated]");
|
||||
expect(archived).not.toContain("\uFFFD");
|
||||
} finally {
|
||||
delete BUILTIN_TOOL_FACTORIES[NAME];
|
||||
}
|
||||
});
|
||||
|
||||
it("recovers a compatibility tool that returns only a complete message", async () => {
|
||||
const NAME = "__complete_message_tool__";
|
||||
const source = `COMPLETE-ONLY-${"z".repeat(100)}-END`;
|
||||
BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({
|
||||
name: NAME,
|
||||
definition,
|
||||
async *execute(_args, ctx) {
|
||||
yield toolCallOutput({
|
||||
output: source,
|
||||
toolCallId: ctx.toolCallId,
|
||||
stopReason: "completed",
|
||||
});
|
||||
},
|
||||
});
|
||||
try {
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: {
|
||||
customTools: [
|
||||
{ name: NAME, description: "complete", permission: "r", maxOutputLength: 20 },
|
||||
],
|
||||
mcpServers: [],
|
||||
},
|
||||
sessionScratchpadDir: path.join(tmp, "scratch"),
|
||||
});
|
||||
const messages = await collect(
|
||||
env.executeTool({
|
||||
toolCall: toolCall({ name: NAME, arguments: "{}", toolCallId: "complete-only" }),
|
||||
}),
|
||||
);
|
||||
const complete = messages[messages.length - 1]!.payload as {
|
||||
output: string;
|
||||
stop_reason?: string;
|
||||
};
|
||||
const savedPath = recoveryPath(complete.output);
|
||||
expect(savedPath).toBeDefined();
|
||||
expect(await readFile(savedPath!, "utf8")).toBe(source);
|
||||
expect(complete.output.startsWith(source.slice(0, 20))).toBe(true);
|
||||
expect(complete.stop_reason).toBe("completed");
|
||||
const streamed = messages
|
||||
.filter(
|
||||
(m) =>
|
||||
(m.payload as { type?: string }).type === "partial_tool_call_output" &&
|
||||
(m.payload as { event_type?: string }).event_type === "delta",
|
||||
)
|
||||
.map((m) => (m.payload as { output?: string }).output ?? "")
|
||||
.join("");
|
||||
expect(streamed).toBe(complete.output);
|
||||
env.dispose();
|
||||
expect(await readFile(savedPath!, "utf8")).toBe(source);
|
||||
} finally {
|
||||
delete BUILTIN_TOOL_FACTORIES[NAME];
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -461,13 +817,14 @@ describe("Environment.executeTool — relaxed tool contract", () => {
|
||||
it("passes origin-tagged nested messages through verbatim, excluded from the tool output", async () => {
|
||||
const NAME = "__forwarding_tool__";
|
||||
const hop = "sess_child";
|
||||
const childOutput = "child result ".repeat(100);
|
||||
BUILTIN_TOOL_FACTORIES[NAME] = (definition) => ({
|
||||
name: NAME,
|
||||
definition,
|
||||
async *execute(_args, ctx) {
|
||||
// Nested forwarding: origin-tagged messages pass through verbatim (a child session's
|
||||
// complete tool_call_output is not folded into the finish either).
|
||||
yield withOrigin(toolCallOutput({ output: "child result", toolCallId: "child_call" }), hop);
|
||||
yield withOrigin(toolCallOutput({ output: childOutput, toolCallId: "child_call" }), hop);
|
||||
yield partialToolCallOutput({
|
||||
eventType: "delta",
|
||||
output: "own output",
|
||||
@@ -476,12 +833,15 @@ describe("Environment.executeTool — relaxed tool contract", () => {
|
||||
},
|
||||
});
|
||||
try {
|
||||
const sessionScratchpadDir = path.join(tmp, "scratch");
|
||||
const truncatedToolOutputRoot = path.join(sessionScratchpadDir, "truncated-tool-output");
|
||||
const env = new Environment({
|
||||
workspaceDir: tmp,
|
||||
toolConfig: {
|
||||
customTools: [{ name: NAME, description: "fwd", permission: "rw" }],
|
||||
customTools: [{ name: NAME, description: "fwd", permission: "rw", maxOutputLength: 10 }],
|
||||
mcpServers: [],
|
||||
},
|
||||
sessionScratchpadDir,
|
||||
});
|
||||
const out = await collect(
|
||||
env.executeTool({
|
||||
@@ -491,7 +851,7 @@ describe("Environment.executeTool — relaxed tool contract", () => {
|
||||
// The forwarded nested message keeps its origin and original payload.
|
||||
const forwarded = out.find((m) => m.origin?.length);
|
||||
expect(forwarded).toBeDefined();
|
||||
expect((forwarded!.payload as { output?: string }).output).toBe("child result");
|
||||
expect((forwarded!.payload as { output?: string }).output).toBe(childOutput);
|
||||
// This tool's own complete output contains only its own deltas, not mixed with the
|
||||
// child session's content.
|
||||
const completes = out.filter(
|
||||
@@ -500,6 +860,9 @@ describe("Environment.executeTool — relaxed tool contract", () => {
|
||||
expect(completes).toHaveLength(1);
|
||||
expect((completes[0]!.payload as { output?: string }).output).toBe("own output");
|
||||
expect((completes[0]!.payload as { stop_reason?: string }).stop_reason).toBe("completed");
|
||||
// The oversized child result belongs to the child's Session. The parent's own
|
||||
// output fits exactly, so the parent must not create a duplicate recovery archive.
|
||||
await expect(access(truncatedToolOutputRoot)).rejects.toThrow();
|
||||
} finally {
|
||||
delete BUILTIN_TOOL_FACTORIES[NAME];
|
||||
}
|
||||
|
||||
@@ -451,6 +451,43 @@ describe("read_file — bounded scan and output budget", () => {
|
||||
expect(text).toContain("sed -n");
|
||||
});
|
||||
|
||||
it.skipIf(process.platform !== "win32")(
|
||||
"accepts the same absolute path in Windows and POSIX spellings",
|
||||
async () => {
|
||||
await writeFile(path.join(tmp, "spellings.txt"), "both spellings work\n");
|
||||
const windowsSpelling = path.join(tmp, "spellings.txt");
|
||||
const posixSpelling = windowsSpelling.replaceAll("\\", "/");
|
||||
const tool = () => createReadFileTool(def(READ_FILE_NAME, "r"));
|
||||
for (const spelling of [windowsSpelling, posixSpelling]) {
|
||||
const { text } = await run(tool(), { file_path: spelling }, tmp);
|
||||
expect(text).toContain("both spellings work");
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
it.skipIf(process.platform !== "win32")(
|
||||
"names the workspace with forward slashes when a file is missing",
|
||||
async () => {
|
||||
const tool = () => createReadFileTool(def(READ_FILE_NAME, "r"));
|
||||
const { text } = await run(tool(), { file_path: "no-such-file.txt" }, tmp);
|
||||
expect(text).toContain("File not found");
|
||||
expect(text).not.toContain("\\");
|
||||
},
|
||||
);
|
||||
|
||||
it("confirms EOF for an unterminated line one byte below the scan cap", async () => {
|
||||
await writeFile(
|
||||
path.join(tmp, "scan-cap-minus-one.txt"),
|
||||
"x".repeat(READ_FILE_SCAN_CAP_BYTES - 1),
|
||||
);
|
||||
const tool = () => createReadFileTool(def(READ_FILE_NAME, "r"));
|
||||
const { result, text } = await run(tool(), { file_path: "scan-cap-minus-one.txt" }, tmp);
|
||||
expect(result?.stopReason).toBeUndefined();
|
||||
expect(text).toContain("[line truncated]");
|
||||
expect(text).not.toContain("file has more than");
|
||||
expect(text).not.toContain("Stopped after scanning");
|
||||
});
|
||||
|
||||
it("reports a lower-bound total when the file outruns the scan cap after the window", async () => {
|
||||
// > 8MB of two-byte lines: the window (1-2000) completes early, counting stops at the cap.
|
||||
await writeFile(path.join(tmp, "long.txt"), "x\n".repeat(READ_FILE_SCAN_CAP_BYTES / 2 + 4096));
|
||||
|
||||
@@ -15,7 +15,9 @@ import {
|
||||
goalFinishedOf,
|
||||
imageUrlMessage,
|
||||
isGoalRoundInput,
|
||||
modelVisiblePath,
|
||||
parseGoalMessage,
|
||||
sessionScratchpadDir,
|
||||
stripConversationMarkers,
|
||||
tokenUsage,
|
||||
userText,
|
||||
@@ -211,8 +213,14 @@ describe("[goal] marker parsing", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("goal paths", () => {
|
||||
it("derives the goal file path from the scratchpad session directory", () => {
|
||||
describe("session scratchpad paths", () => {
|
||||
it("derives one Session's scratchpad directory", () => {
|
||||
expect(sessionScratchpadDir("/root", "p", "a", "s1")).toBe(
|
||||
path.join("/root", "p", "agents", "a", "scratchpad", "s1"),
|
||||
);
|
||||
});
|
||||
|
||||
it("derives the goal file path from the same Session scratchpad", () => {
|
||||
expect(goalFilePath("/root", "p", "a", "s1")).toBe(
|
||||
path.join("/root", "p", "agents", "a", "scratchpad", "s1", "GOAL.yaml"),
|
||||
);
|
||||
@@ -474,7 +482,7 @@ describe("Session.runGoal input", () => {
|
||||
// The picture is on disk, and both rounds point at that same file.
|
||||
const saved = await fs.readdir(path.join(dir, "scratchpad", "session-1"));
|
||||
expect(saved).toHaveLength(1);
|
||||
const line = `[attached image: ${path.join(dir, "scratchpad", "session-1", saved[0]!)}]`;
|
||||
const line = `[attached image: ${modelVisiblePath(path.join(dir, "scratchpad", "session-1", saved[0]!))}]`;
|
||||
for (const text of rounds) expect(text).toContain(line);
|
||||
// Round 2 re-injects the objective alone, which is where the line matters most: it
|
||||
// survives because stripLeadingMarkerBlocks only removes leading blocks, and the fold
|
||||
|
||||
@@ -11,6 +11,7 @@ import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import { mkdtemp, readFile, readdir, rm, writeFile } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import { modelVisiblePath } from "../src/internal/model-visible-path.js";
|
||||
import { appendAttachmentLines, imagesToScratchpadPaths } from "../src/internal/session-support.js";
|
||||
import { Session } from "../src/index.js";
|
||||
import type {
|
||||
@@ -69,7 +70,7 @@ describe("imagesToScratchpadPaths", () => {
|
||||
const files = await readdir(dir);
|
||||
expect(files).toHaveLength(2);
|
||||
for (const f of paths) {
|
||||
expect(path.dirname(f)).toBe(dir);
|
||||
expect(path.dirname(f)).toBe(modelVisiblePath(dir));
|
||||
expect(path.basename(f)).toMatch(/^upload-[0-9a-f]{8}\.png$/);
|
||||
expect(await readFile(f)).toEqual(PNG_1X1);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { modelVisiblePath } from "../src/internal/model-visible-path.js";
|
||||
|
||||
describe("modelVisiblePath", () => {
|
||||
it.skipIf(process.platform === "win32")(
|
||||
"passes POSIX paths through, including backslash filename characters",
|
||||
() => {
|
||||
expect(modelVisiblePath("/tmp/a path/report.log")).toBe("/tmp/a path/report.log");
|
||||
expect(modelVisiblePath("/tmp/we\\ird.log")).toBe("/tmp/we\\ird.log");
|
||||
},
|
||||
);
|
||||
|
||||
it.skipIf(process.platform !== "win32")(
|
||||
"spells ordinary Windows and UNC paths with forward slashes",
|
||||
() => {
|
||||
expect(modelVisiblePath("C:\\Users\\x\\a path\\report.log")).toBe(
|
||||
"C:/Users/x/a path/report.log",
|
||||
);
|
||||
expect(modelVisiblePath("\\\\server\\share\\report.log")).toBe("//server/share/report.log");
|
||||
},
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,209 @@
|
||||
import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import path from "node:path";
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import {
|
||||
TRUNCATED_TOOL_OUTPUT_FILE_LIMIT_BYTES,
|
||||
TruncatedToolOutputArchive,
|
||||
} from "../src/environment/truncated-tool-output-archive.js";
|
||||
|
||||
let tmp: string;
|
||||
|
||||
beforeEach(async () => {
|
||||
tmp = await mkdtemp(path.join(tmpdir(), "penguin-truncated-tool-output-"));
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
await rm(tmp, { recursive: true, force: true, maxRetries: 3, retryDelay: 50 });
|
||||
});
|
||||
|
||||
describe("TruncatedToolOutputArchive", () => {
|
||||
it("pins the production per-call budget", () => {
|
||||
expect(TRUNCATED_TOOL_OUTPUT_FILE_LIMIT_BYTES).toBe(8 * 1024 * 1024 - 1);
|
||||
});
|
||||
|
||||
it("writes an exact UTF-8 log with private permissions and leaves it Session-owned", async () => {
|
||||
const ordinaryDir = path.join(tmp, "ordinary-session-dir");
|
||||
const sessionDir = path.join(tmp, "archive-session-dir");
|
||||
await mkdir(ordinaryDir);
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(sessionDir, "output"),
|
||||
fileLimitBytes: 128,
|
||||
});
|
||||
const capture = archive.startCapture();
|
||||
capture.append("hello ");
|
||||
capture.append("企鹅");
|
||||
// Split one surrogate pair across deltas: the recovery file must reconstruct the same text
|
||||
// instead of serializing two replacement characters.
|
||||
capture.append("\ud83d");
|
||||
capture.append("\udc27");
|
||||
// If another high surrogate arrives first, only the old one is known to be lone; the new one
|
||||
// must remain pending so it can still pair with the following low surrogate.
|
||||
capture.append("|");
|
||||
capture.append("\ud83d");
|
||||
capture.append("\ud83d");
|
||||
capture.append("\udc27");
|
||||
|
||||
const retained = capture as unknown as { exactChunks: Buffer[] };
|
||||
expect(retained.exactChunks.length).toBeGreaterThan(0);
|
||||
expect(retained.exactChunks.every((chunk) => Buffer.isBuffer(chunk))).toBe(true);
|
||||
|
||||
const saved = await capture.save("exec_command", "call/private");
|
||||
expect(saved.status).toBe("saved");
|
||||
if (saved.status !== "saved") throw new Error("expected saved output");
|
||||
expect(saved.archiveTruncated).toBe(false);
|
||||
expect(await readFile(saved.path, "utf8")).toBe("hello 企鹅🐧|�🐧");
|
||||
if (process.platform !== "win32") {
|
||||
expect((await stat(saved.path)).mode & 0o777).toBe(0o600);
|
||||
expect((await stat(path.dirname(saved.path))).mode & 0o777).toBe(0o700);
|
||||
expect((await stat(sessionDir)).mode & 0o777).toBe((await stat(ordinaryDir)).mode & 0o777);
|
||||
}
|
||||
|
||||
// The archive owns no Task/runtime cleanup. The host removes this whole directory through
|
||||
// the existing Session scratchpad deletion path.
|
||||
expect(await readFile(saved.path, "utf8")).toContain("hello");
|
||||
});
|
||||
|
||||
it("keeps UTF-8-safe head and tail windows when one archive exceeds its file budget", async () => {
|
||||
const fileLimitBytes = 96;
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(tmp, "output"),
|
||||
fileLimitBytes,
|
||||
});
|
||||
const capture = archive.startCapture();
|
||||
const source = `HEAD-${"企鹅🐧".repeat(80)}-TAIL`;
|
||||
// Exercise incremental rolling-tail updates rather than one monolithic append.
|
||||
capture.append(source.slice(0, 70));
|
||||
capture.append(source.slice(70, 210));
|
||||
capture.append(source.slice(210));
|
||||
|
||||
const saved = await capture.save("describe/image", "unicode-call");
|
||||
expect(saved.status).toBe("saved");
|
||||
if (saved.status !== "saved") throw new Error("expected saved output");
|
||||
const archived = await readFile(saved.path, "utf8");
|
||||
expect(saved.archiveTruncated).toBe(true);
|
||||
expect(Buffer.byteLength(archived, "utf8")).toBeLessThanOrEqual(fileLimitBytes);
|
||||
expect(archived).toContain("HEAD-");
|
||||
expect(archived).toContain("-TAIL");
|
||||
expect(archived).toContain("[archive middle truncated]");
|
||||
expect(archived).not.toContain("\uFFFD");
|
||||
});
|
||||
|
||||
it("does not retain a huge delta's backing buffer after reducing it to bounded windows", async () => {
|
||||
const fileLimitBytes = 96;
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(tmp, "output"),
|
||||
fileLimitBytes,
|
||||
});
|
||||
const capture = archive.startCapture();
|
||||
capture.append(`HEAD-${"x".repeat(16 * 1024 * 1024)}-TAIL`);
|
||||
|
||||
const retained = capture as unknown as {
|
||||
exactChunks: Buffer[];
|
||||
head: Buffer;
|
||||
tail: Buffer;
|
||||
tailLength: number;
|
||||
};
|
||||
expect(retained.exactChunks).toEqual([]);
|
||||
expect(retained.head.length + retained.tailLength).toBeLessThanOrEqual(fileLimitBytes);
|
||||
// A short subarray of the original 16 MiB Buffer would pass the length assertion while
|
||||
// still pinning the entire ArrayBuffer. Node may use its fixed small-buffer pool for this
|
||||
// tiny test override, but neither result may retain the huge source.
|
||||
const largestBoundedBackingStore = Math.max(fileLimitBytes, Buffer.poolSize);
|
||||
expect(retained.head.buffer.byteLength).toBeLessThanOrEqual(largestBoundedBackingStore);
|
||||
expect(retained.tail.buffer.byteLength).toBeLessThanOrEqual(largestBoundedBackingStore);
|
||||
|
||||
const saved = await capture.save("tool", "huge-single-delta");
|
||||
expect(saved.status).toBe("saved");
|
||||
if (saved.status !== "saved") throw new Error("expected saved output");
|
||||
const archived = await readFile(saved.path, "utf8");
|
||||
expect(Buffer.byteLength(archived, "utf8")).toBeLessThanOrEqual(fileLimitBytes);
|
||||
expect(archived).toContain("HEAD-");
|
||||
expect(archived).toContain("-TAIL");
|
||||
});
|
||||
|
||||
it("reuses fixed tail storage across many small post-promotion deltas", async () => {
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(tmp, "output"),
|
||||
fileLimitBytes: 128,
|
||||
});
|
||||
const capture = archive.startCapture();
|
||||
capture.append("H".repeat(256));
|
||||
const retained = capture as unknown as {
|
||||
tail: Buffer;
|
||||
tailLength: number;
|
||||
};
|
||||
const tailStorage = retained.tail;
|
||||
// Vary encoded widths and split emoji pairs across deltas so ring wrap-around must preserve
|
||||
// the same UTF-8 and surrogate semantics as the exact phase.
|
||||
const pattern = ["企", "鹅", "\ud83d", "\udc27", "|"];
|
||||
const chunks = Array.from({ length: 4096 }, (_, index) => pattern[index % pattern.length]!);
|
||||
for (const chunk of chunks) capture.append(chunk);
|
||||
|
||||
expect(retained.tail).toBe(tailStorage);
|
||||
expect(retained.tailLength).toBeLessThanOrEqual(tailStorage.length);
|
||||
|
||||
const saved = await capture.save("tool", "many-small-deltas");
|
||||
expect(saved.status).toBe("saved");
|
||||
if (saved.status !== "saved") throw new Error("expected saved output");
|
||||
const archived = await readFile(saved.path, "utf8");
|
||||
const archivedTail = archived.split("[archive middle truncated]\n")[1] ?? "";
|
||||
expect(chunks.join("").endsWith(archivedTail)).toBe(true);
|
||||
expect(archived).not.toContain("\uFFFD");
|
||||
});
|
||||
|
||||
it("allows independent captures without imposing an aggregate Session budget", async () => {
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(tmp, "output"),
|
||||
fileLimitBytes: 64,
|
||||
});
|
||||
const captures = [archive.startCapture(), archive.startCapture(), archive.startCapture()];
|
||||
captures.forEach((capture, index) => capture.append(`output-${index}`));
|
||||
|
||||
const saved = await Promise.all(
|
||||
captures.map((capture, index) => capture.save("tool", `call-${index}`)),
|
||||
);
|
||||
expect(saved.every((result) => result.status === "saved")).toBe(true);
|
||||
});
|
||||
|
||||
it("reports a write failure without throwing", async () => {
|
||||
const rootFile = path.join(tmp, "not-a-directory");
|
||||
await writeFile(rootFile, "occupied", "utf8");
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: rootFile,
|
||||
fileLimitBytes: 64,
|
||||
});
|
||||
const capture = archive.startCapture();
|
||||
capture.append("output");
|
||||
const stderr: string[] = [];
|
||||
const stderrSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
|
||||
stderr.push(String(chunk));
|
||||
return true;
|
||||
});
|
||||
try {
|
||||
const result = await capture.save("tool", "call");
|
||||
expect(result.status).toBe("failed");
|
||||
if (result.status !== "failed") throw new Error("expected failed output");
|
||||
expect(typeof result.code).toBe("string");
|
||||
expect(stderr.join("")).toContain(
|
||||
`[penguin] tool "tool" truncated output archive write failed (${result.code}).`,
|
||||
);
|
||||
} finally {
|
||||
stderrSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
|
||||
it("distinguishes a repeated save from an archive write failure", async () => {
|
||||
const archive = new TruncatedToolOutputArchive({
|
||||
rootDir: path.join(tmp, "output"),
|
||||
fileLimitBytes: 64,
|
||||
});
|
||||
const capture = archive.startCapture();
|
||||
capture.append("output");
|
||||
expect((await capture.save("tool", "call")).status).toBe("saved");
|
||||
await expect(capture.save("tool", "call")).resolves.toEqual({
|
||||
status: "failed",
|
||||
code: "ALREADY_SAVED",
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -161,6 +161,8 @@ An existing Agent always runs with its on-disk config verbatim — newer code de
|
||||
|
||||
`{{PROJECT_DIR}}` is surfaced to the model as the **App Data Dir**: PenguinHarness's application data root, holding every Agent's data files (`agents/<agent_id>/…`) and the project-level data — deliberately not described as a project or task directory, so the model does not mistake it for the task's working directory (`CWD`).
|
||||
|
||||
On Windows, `{{PROJECT_DIR}}` and `{{CWD}}` are injected with forward slashes — like every other path core composes for the model (attachment lines, the goal-file line, truncated-output recovery paths). The model re-emits these spellings into JSON tool arguments and shell commands; forward slashes are accepted by Node's fs APIs and the package's (Git) Bash tool shell, and avoid JSON backslash-escaping mistakes.
|
||||
|
||||
`agent_state/AGENTS.md` is the developer-editable instruction file, injected via `{{AGENTS_MD}}` and empty by default — it is also the file an optimizer edits most (see [Self-Improvement](/self-improvement)).
|
||||
|
||||
## Vault
|
||||
|
||||
@@ -161,6 +161,8 @@ compaction:
|
||||
|
||||
`{{PROJECT_DIR}}` 在提示词中以 **App Data Dir** 名义暴露给模型:PenguinHarness 的应用数据根目录,存放全部 Agent 的数据文件(`agents/<agent_id>/…`)与 Project 级数据——特意不以 Project/任务目录的口径描述,避免模型将其误认为本次任务的工作目录(`CWD`)。
|
||||
|
||||
Windows 上注入的 `{{PROJECT_DIR}}` 与 `{{CWD}}` 统一使用正斜杠——与 core 产出的其他模型可见路径(附件行、Goal file 行、截断输出 recovery 路径)同一拼写。模型会把这些拼写原样带入 JSON 工具参数和 Shell 命令;正斜杠被 Node 的 fs API 与包内 (Git) Bash 工具 Shell 接受,也避免 JSON 反斜杠转义出错。
|
||||
|
||||
`agent_state/AGENTS.md` 是开发者可编辑的指令文件,经 `{{AGENTS_MD}}` 注入系统提示词,缺省为空——它也是优化器最常改动的文件(见[自我进化](/self-improvement))。
|
||||
|
||||
## Vault
|
||||
|
||||
@@ -110,7 +110,7 @@ interface EnvironmentInterface {
|
||||
}
|
||||
```
|
||||
|
||||
`executeTool` yields `partial_tool_call_output` fragments and ends with exactly one complete `tool_call_output`; `origin`-tagged nested messages (e.g. forwarded by `run_subagent`) pass through unchanged. Rendering is explicitly not this interface's concern — streaming rendering belongs to the CLI / Web front ends.
|
||||
`executeTool` yields `partial_tool_call_output` fragments and ends with exactly one complete `tool_call_output`; `origin`-tagged nested messages (e.g. forwarded by `run_subagent`) pass through unchanged. The built-in Environment can keep truncated text in the Session scratchpad without exposing storage lifecycle hooks through this public interface. Its model-visible recovery path is a plain absolute path; on Windows it is written with forward slashes, which Node's fs APIs and the package's (Git) Bash tool shell both accept, so the same spelling works as a `read_file` argument and inside shell commands. Rendering is explicitly not this interface's concern — streaming rendering belongs to the CLI / Web front ends.
|
||||
|
||||
### ToolExecutionRequest and EnvironmentConfig
|
||||
|
||||
@@ -124,6 +124,7 @@ interface ToolExecutionRequest {
|
||||
interface EnvironmentConfig {
|
||||
workspaceDir: string;
|
||||
toolConfig: ToolConfig; // { customTools: ToolDefinitionConfig[]; mcpServers: MCPServerConfig[] }
|
||||
sessionScratchpadDir?: string; // this Session's scratchpad (scratchpad/<sessionId>); enables truncated-output recovery
|
||||
services?: EnvironmentServices; // runtime services injected into individual tools
|
||||
vault?: Record<string, string>; // Vault env vars, injected into exec_command / input_command subprocesses
|
||||
}
|
||||
@@ -141,6 +142,18 @@ interface MCPServerConfig {
|
||||
}
|
||||
```
|
||||
|
||||
`Agent.createSession()` and `resumeSession()` pass the Session scratchpad directory
|
||||
automatically. A standalone embedder that owns a stable per-Session directory opts in by
|
||||
supplying it — no archive-specific type is exposed:
|
||||
|
||||
```ts
|
||||
const environment = new Environment({
|
||||
workspaceDir,
|
||||
toolConfig,
|
||||
sessionScratchpadDir, // e.g. <dataRoot>/<project>/agents/<agent>/scratchpad/<sessionId>
|
||||
});
|
||||
```
|
||||
|
||||
### The inner tool contract: BuiltinTool
|
||||
|
||||
Inside the Environment, an individual tool follows a deliberately narrower contract ("loose tool, strict framework"):
|
||||
|
||||
@@ -110,7 +110,7 @@ interface EnvironmentInterface {
|
||||
}
|
||||
```
|
||||
|
||||
`executeTool` 逐条产出 `partial_tool_call_output`,并以恰好一条完整 `tool_call_output` 收尾;带 `origin` 的嵌套消息(如 `run_subagent` 转发的子 Session 消息)原样透传。渲染不是本接口的职责——流式渲染由 CLI / Web 前端完成。
|
||||
`executeTool` 逐条产出 `partial_tool_call_output`,并以恰好一条完整 `tool_call_output` 收尾;带 `origin` 的嵌套消息(如 `run_subagent` 转发的子 Session 消息)原样透传。内置 Environment 可以把被截断文本保存在 Session scratchpad 中,无需在此公共接口暴露存储生命周期钩子。其模型可见 recovery 路径是普通绝对路径;Windows 上统一写成正斜杠——Node 的 fs API 与包内 (Git) Bash 工具 Shell 都接受这种写法,同一拼写既可直接作 `read_file` 参数、也可用于 Shell 命令。渲染不是本接口的职责——流式渲染由 CLI / Web 前端完成。
|
||||
|
||||
### ToolExecutionRequest 与 EnvironmentConfig
|
||||
|
||||
@@ -124,6 +124,7 @@ interface ToolExecutionRequest {
|
||||
interface EnvironmentConfig {
|
||||
workspaceDir: string;
|
||||
toolConfig: ToolConfig; // { customTools: ToolDefinitionConfig[]; mcpServers: MCPServerConfig[] }
|
||||
sessionScratchpadDir?: string; // 本 Session 的 scratchpad(scratchpad/<sessionId>),提供后启用截断输出恢复
|
||||
services?: EnvironmentServices; // 注入给个别工具的运行时服务
|
||||
vault?: Record<string, string>; // Vault 环境变量,注入 exec_command / input_command 子进程
|
||||
}
|
||||
@@ -141,6 +142,17 @@ interface MCPServerConfig {
|
||||
}
|
||||
```
|
||||
|
||||
`Agent.createSession()` 与 `resumeSession()` 会自动传入 Session scratchpad 目录。自行管理稳定
|
||||
per-Session 目录的独立 embedder 只需提供该目录即可启用,不暴露归档专用类型:
|
||||
|
||||
```ts
|
||||
const environment = new Environment({
|
||||
workspaceDir,
|
||||
toolConfig,
|
||||
sessionScratchpadDir, // 例如 <dataRoot>/<project>/agents/<agent>/scratchpad/<sessionId>
|
||||
});
|
||||
```
|
||||
|
||||
### 内层工具契约:BuiltinTool
|
||||
|
||||
Environment 之内,单个工具遵循更窄的契约(「松工具、紧框架」):
|
||||
|
||||
@@ -70,6 +70,8 @@ The head of a Trace (illustrative; one OmniMessage envelope per line):
|
||||
{"timestamp":"…","type":"event_msg","payload":{"type":"token_usage","session":{…},"request":{…}}}
|
||||
```
|
||||
|
||||
When tool output exceeds `maxOutputLength`, Trace records the same bounded head, truncation marker, and absolute Session recovery path seen by Web/CLI and the model; it does not separately duplicate the archived text. The path exposes the host data-root layout but remains valid across Tasks and Session resume because the unredacted recovery file lives in that Session's scratchpad. The existing explicit Session-deletion path removes the scratchpad and recovery file together. Trace replay therefore faithfully restores both what the model saw and a usable pointer for later follow-up.
|
||||
|
||||
## Session recovery
|
||||
|
||||
The Trace is the single source of truth for recovery — there is no separate session database to keep in sync. `resumeSession` works as follows:
|
||||
|
||||
@@ -68,6 +68,8 @@ Trace 是 append-only 的 JSON Lines 文件,每行一个 OmniMessage 信封(
|
||||
{"timestamp":"…","type":"event_msg","payload":{"type":"token_usage","session":{…},"request":{…}}}
|
||||
```
|
||||
|
||||
工具输出超过 `maxOutputLength` 时,Trace 与 Web/CLI、模型一样只记录有界头部、截断提示和绝对 Session recovery 路径,不重复保存归档正文。该路径会暴露宿主的数据根目录布局;未经脱敏的 recovery 文件位于该 Session 的 scratchpad,因此路径跨 Task 和 Session 恢复保持有效。用户明确删除 Session 时,现有删除路径会连同 scratchpad 和 recovery 文件一起清理。因而 Trace 重放既忠实恢复「模型当时看到了什么」,也为后续追问保留可用指针。
|
||||
|
||||
## Session 恢复
|
||||
|
||||
Trace 是恢复的唯一事实来源,没有独立的会话数据库需要与之对齐。`resumeSession` 的流程:
|
||||
|
||||
@@ -45,6 +45,16 @@ A tool only yields incremental `partial_tool_call_output` deltas; the Environmen
|
||||
|
||||
Tools and the Environment never throw into the engine: errors collapse into `tool_call_output` messages the model can read and react to. See the [OmniMessage Protocol](/omni-message) for message structure.
|
||||
|
||||
### Recovering oversized output
|
||||
|
||||
When tool text in an Agent Session exceeds `maxOutputLength`, the model and Web/CLI still receive the same head window, truncation marker, and terminal marker, and the streaming invariant that user-visible output equals model-visible output does not change. Environment also appends a short archive status/path note outside that visible-output cap and saves a Session-owned recovery file. The file is exact within the per-call archive budget and otherwise contains bounded head/tail windows. This is the complete text **received by Environment**: a producer such as a command or subagent session may already have replaced overflow with an `[..., N chars of earlier output dropped ...]` marker in its own bounded unread buffer, and the downstream archive cannot recover text lost before that point.
|
||||
|
||||
The Agent can inspect ordinary multiline archives with the existing `read_file` (`offset` / `limit`). For byte tails or very long lines, it must construct a targeted shell command such as `rg` / `tail`; no dedicated retrieval tool is added. The note carries a plain absolute path, always the last element inside the bracket. On Windows it is written with forward slashes: `exec_command` runs through (Git) Bash and Node's fs APIs accept them, so one spelling works in JSON tool arguments and shell commands alike; POSIX paths pass through unchanged, and Session paths are ordinary absolute paths (never `\\?\`-prefixed), so the separator swap is lossless. As with any path, quote it inside shell commands when it contains spaces. The same spelling rule covers every path core composes for the model — the system prompt's App Data Dir / CWD lines, `[attached image/file: …]` lines and the goal-file line (`modelVisiblePath` in the SDK).
|
||||
|
||||
Recovery files live under the Session's `scratchpad/<session-id>/truncated-tool-output/`, are created only after actual truncation, and use private permissions where the platform supports them. One call stores at most 8 MiB (the production byte limit is one byte lower so `read_file` remains below its 8 MiB scan cap); larger output keeps bounded head/tail windows in the file with an explicit middle-gap marker. The limit is per call only: a Session has no aggregate archive byte or file-count quota, and concurrent captures independently retain up to one call's budget. Files remain readable across Tasks, runtime disposal, and Session resume until explicit Session deletion removes the entire scratchpad; no separate archive cleanup lifecycle is added.
|
||||
|
||||
Recovery files contain the unredacted tool text received by Environment. Accidentally reading credentials or other sensitive data can therefore increase local at-rest retention from the visible head window to the archive budget. Trace does not duplicate those bytes, but it records the same absolute Session path shown to the model and Web/CLI, exposing the host's data-root layout. Archive-write failure never changes the original tool's `stop_reason`; the visible note and stderr warning carry only a short error code (and stderr's tool name), not the path or raw error message.
|
||||
|
||||
## Configuration fields
|
||||
|
||||
Each tool is described by one `ToolDefinitionConfig`:
|
||||
|
||||
@@ -45,6 +45,16 @@ interface ToolResult {
|
||||
|
||||
工具与 Environment 从不向引擎抛异常:错误一律折叠为 `tool_call_output` 消息,交给模型阅读并调整下一步。消息结构见 [OmniMessage 协议](/omni-message)。
|
||||
|
||||
### 过长输出恢复
|
||||
|
||||
Agent Session 中的工具文本超过 `maxOutputLength` 时,模型与 Web/CLI 仍只收到相同的头部窗口、截断提示与终止标记,「用户所见 = 模型所见」的流式契约也保持不变。Environment 还会在该可见输出上限之外追加一条简短的归档状态/路径 note,并保存归该 Session 所有的 recovery 文件:单次归档预算内保存完整文本,超出预算则保存有界头尾。这里的「完整」特指 **Environment 实际收到的文本**:命令或子 Agent Session 等生产者可能已在自身的有界未读缓冲区中用 `[..., N chars of earlier output dropped ...]` 标记替换溢出内容,下游归档无法恢复在此之前已经丢失的原文。
|
||||
|
||||
普通多行归档可用现有 `read_file`(`offset` / `limit`)查看;若要读取字节级尾部或超长单行,Agent 必须自行构造定向的 `rg` / `tail` 等 Shell 命令,不新增专用读取工具。note 中的路径是普通绝对路径,恒为括号内最后一个元素。Windows 上统一写成正斜杠:`exec_command` 经 (Git) Bash 执行、Node 的 fs API 也接受正斜杠,同一拼写在 JSON 工具参数与 Shell 命令中通用;POSIX 路径原样透传,且 Session 路径都是普通绝对路径(不会带 `\\?\` 前缀),分隔符替换无损。含空格的路径在 Shell 命令中照常引用即可。同一拼写规则覆盖 core 产出给模型的全部路径——系统提示词的 App Data Dir / CWD 行、`[attached image/file: …]` 行与 Goal file 行(SDK 中的 `modelVisiblePath`)。
|
||||
|
||||
Recovery 文件位于该 Session 的 `scratchpad/<session-id>/truncated-tool-output/`,仅在确实发生截断时创建;平台支持时使用仅当前用户可读写的私有权限。单次调用最多保存 8 MiB(生产字节上限少 1 byte,以保持低于 `read_file` 的 8 MiB 扫描上限);更大的输出在文件中保留有界头尾并写明中间被截。该限制仅针对单次调用:一个 Session 没有归档总字节数或文件数配额,并发捕获也各自最多保留一份单调用预算。文件跨 Task、运行时释放和 Session 恢复保持可读,直到用户明确删除 Session 时由现有路径连同整个 scratchpad 一起移除;不新增单独的归档清理生命周期。
|
||||
|
||||
Recovery 文件保存 Environment 收到的未经脱敏的工具文本。误读凭据或其他敏感数据会使本地静态留存量从可见头部扩大到归档预算。Trace 不重复保存这些正文,但会记录模型与 Web/CLI 看到的同一个绝对 Session 路径,因此会暴露宿主的数据根目录布局。归档写入失败不改变原工具的 `stop_reason`;双方可见的 note 与 stderr 警告只携带简短错误码(stderr 另含工具名),不携带路径或原始错误消息。
|
||||
|
||||
## 配置字段
|
||||
|
||||
每个工具由一条 `ToolDefinitionConfig` 描述:
|
||||
|
||||
@@ -16,7 +16,11 @@
|
||||
import fs from "node:fs/promises";
|
||||
import path from "node:path";
|
||||
import { randomBytes } from "node:crypto";
|
||||
import { appendAttachmentLines, attachedFileLine } from "@prismshadow/penguin-core";
|
||||
import {
|
||||
appendAttachmentLines,
|
||||
attachedFileLine,
|
||||
modelVisiblePath,
|
||||
} from "@prismshadow/penguin-core";
|
||||
import type { OmniMessage } from "@prismshadow/penguin-core";
|
||||
import { HttpError } from "../http/errors.js";
|
||||
import { badRequest } from "../http/validate.js";
|
||||
@@ -293,5 +297,11 @@ export async function attachFilesToInput(
|
||||
await removeAttachments(written);
|
||||
throw err;
|
||||
}
|
||||
return { input: appendAttachmentLines(messages, written.map(attachedFileLine)), written };
|
||||
return {
|
||||
input: appendAttachmentLines(
|
||||
messages,
|
||||
written.map((filePath) => attachedFileLine(modelVisiblePath(filePath))),
|
||||
),
|
||||
written,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
||||
import {
|
||||
assistantText,
|
||||
buildHandoffMessage,
|
||||
modelVisiblePath,
|
||||
parseHandoffMessage,
|
||||
scratchpadDir,
|
||||
} from "@prismshadow/penguin-core";
|
||||
@@ -121,7 +122,8 @@ describe("task input file attachments", () => {
|
||||
const marker = /\[attached file: (.+)\]/.exec(text);
|
||||
expect(marker).not.toBeNull();
|
||||
const filePath = marker![1]!;
|
||||
expect(filePath).toBe(path.join(dir, "report.pdf"));
|
||||
// Marker lines carry the model-visible spelling (forward slashes on Windows).
|
||||
expect(filePath).toBe(modelVisiblePath(path.join(dir, "report.pdf")));
|
||||
expect(await fs.readFile(filePath, "utf8")).toBe("PDF-BYTES");
|
||||
// The line trails the user's own text — it must not replace or reframe the message.
|
||||
expect(text.startsWith("look at this")).toBe(true);
|
||||
@@ -141,7 +143,7 @@ describe("task input file attachments", () => {
|
||||
|
||||
const paths = [...promptText(runs[0]!).matchAll(/\[attached file: (.+)\]/g)].map((m) => m[1]!);
|
||||
expect(paths).toHaveLength(2);
|
||||
expect(paths[0]).toBe(path.join(dir, "notes.txt"));
|
||||
expect(paths[0]).toBe(modelVisiblePath(path.join(dir, "notes.txt")));
|
||||
// The second upload gets a random suffix rather than clobbering the first.
|
||||
expect(paths[1]).not.toBe(paths[0]);
|
||||
expect(path.basename(paths[1]!)).toMatch(/^notes-[0-9a-f]{6}\.txt$/);
|
||||
@@ -287,7 +289,7 @@ describe("task input file attachments", () => {
|
||||
expect(texts).toHaveLength(2);
|
||||
expect(texts[0]).toBe(block);
|
||||
expect(parseHandoffMessage(texts[0]!)?.agentId).toBe("alpha");
|
||||
expect(texts[1]).toBe(`[attached file: ${path.join(dir, "notes.txt")}]`);
|
||||
expect(texts[1]).toBe(`[attached file: ${modelVisiblePath(path.join(dir, "notes.txt"))}]`);
|
||||
});
|
||||
|
||||
it("more than the per-request file count is a 413 and writes nothing", async () => {
|
||||
|
||||
@@ -87,15 +87,17 @@ const MAX_PATH_LEN = 512;
|
||||
* the Workspace root (used for file-card display and stat lookups). Returns
|
||||
* null (no card rendered) for anything that can't be resolved into the
|
||||
* current Workspace:
|
||||
* - An absolute path is stripped only when prefixed with
|
||||
* `${workspace}${sep}` (assistants commonly report absolute paths); if it
|
||||
* equals the workspace itself or the prefix doesn't match → null. A
|
||||
* Windows deployment's Workspace (core supports win32) uses backslash
|
||||
* paths: the prefix is joined with its own separator, and the stripped
|
||||
* relative segment is normalized to "/" (the browser-side directory
|
||||
* navigation splits on "/"). Conversion only happens on a matched Windows
|
||||
* prefix — backslash is a legal character in POSIX filenames, so no
|
||||
* global replacement is done;
|
||||
* - An absolute path is stripped only when prefixed with the Workspace
|
||||
* (assistants commonly report absolute paths); if it equals the workspace
|
||||
* itself or the prefix doesn't match → null. A Windows deployment's
|
||||
* Workspace (core supports win32) may be reported with backslashes
|
||||
* (path.join) while the assistant spells the same path with forward
|
||||
* slashes (core's model-visible spelling) or mixes both — so when the
|
||||
* Workspace itself looks like a Windows path, both sides are compared on
|
||||
* "/" with a case-insensitive drive letter, and the stripped relative
|
||||
* segment uses "/" (the browser-side directory navigation splits on
|
||||
* "/"). A POSIX Workspace never converts anything — backslash is a legal
|
||||
* character in POSIX filenames, so no global replacement is done;
|
||||
* - A path starting with `~` (home directory) can't be resolved → null;
|
||||
* - A relative path is lexically normalized by splitting on "/": drop "."
|
||||
* and empty segments, pop the stack on "..", and return null if popping
|
||||
@@ -106,14 +108,19 @@ export function toWorkspaceRelative(path: string, workspace: string | null): str
|
||||
if (s.length === 0 || s.length > MAX_PATH_LEN) return null;
|
||||
if (s.startsWith("~")) return null;
|
||||
const ws = workspace !== null && workspace.length > 0 ? workspace : null;
|
||||
const winWs = ws?.includes("\\") ?? false;
|
||||
const sep = winWs ? "\\" : "/";
|
||||
const absolute = s.startsWith("/") || (winWs && (/^[A-Za-z]:/.test(s) || s.startsWith("\\")));
|
||||
let rel = s;
|
||||
const winWs = ws !== null && (ws.includes("\\") || /^[A-Za-z]:/.test(ws));
|
||||
const normalize = (p: string): string => {
|
||||
if (!winWs) return p;
|
||||
const slashed = p.replaceAll("\\", "/");
|
||||
return /^[A-Za-z]:/.test(slashed) ? slashed[0]!.toLowerCase() + slashed.slice(1) : slashed;
|
||||
};
|
||||
const normalizedInput = normalize(s);
|
||||
const normalizedWs = ws === null ? null : normalize(ws);
|
||||
const absolute = normalizedInput.startsWith("/") || (winWs && /^[a-z]:/.test(normalizedInput));
|
||||
let rel = normalizedInput;
|
||||
if (absolute) {
|
||||
if (ws === null || !s.startsWith(`${ws}${sep}`)) return null;
|
||||
rel = s.slice(ws.length + sep.length);
|
||||
if (sep === "\\") rel = rel.replaceAll("\\", "/");
|
||||
if (normalizedWs === null || !normalizedInput.startsWith(`${normalizedWs}/`)) return null;
|
||||
rel = normalizedInput.slice(normalizedWs.length + 1);
|
||||
}
|
||||
const stack: string[] = [];
|
||||
for (const seg of rel.split("/")) {
|
||||
|
||||
@@ -80,6 +80,24 @@ describe("toWorkspaceRelative", () => {
|
||||
expect(toWorkspaceRelative("C:\\Users\\me\\ws", win)).toBe(null);
|
||||
});
|
||||
|
||||
it("Windows Workspace: forward-slash and mixed spellings of the same path also match", () => {
|
||||
const win = "C:\\Users\\me\\ws";
|
||||
// Core's model-visible spelling uses forward slashes on Windows; the card must still strip.
|
||||
expect(toWorkspaceRelative("C:/Users/me/ws/sub/a.txt", win)).toBe("sub/a.txt");
|
||||
expect(toWorkspaceRelative("C:/Users/me/ws\\sub/a.txt", win)).toBe("sub/a.txt");
|
||||
// Drive letters are case-insensitive on Windows; the rest of the path is not.
|
||||
expect(toWorkspaceRelative("c:/Users/me/ws/sub/a.txt", win)).toBe("sub/a.txt");
|
||||
expect(toWorkspaceRelative("C:/Users/ME/ws/sub/a.txt", win)).toBe(null);
|
||||
// A forward-slash Workspace value matches a backslash assistant path too.
|
||||
expect(toWorkspaceRelative("C:\\Users\\me\\ws\\sub\\a.txt", "C:/Users/me/ws")).toBe(
|
||||
"sub/a.txt",
|
||||
);
|
||||
});
|
||||
|
||||
it("Windows Workspace: relative backslash paths split into segments", () => {
|
||||
expect(toWorkspaceRelative("dir\\name.txt", "C:\\Users\\me\\ws")).toBe("dir/name.txt");
|
||||
});
|
||||
|
||||
it("backslashes in POSIX filenames are not globally replaced (converted only on a Windows prefix match)", () => {
|
||||
expect(toWorkspaceRelative("dir\\name.txt", WS)).toBe("dir\\name.txt");
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user