Initialize repository with harness code and assets

Initial import of all source code, config, and README assets: the
packages workspace (cli, core, server, web, docs, landing, skills),
build scripts, tooling config, and CI workflows.

Includes the data-layout revision made on this branch: the local data
root defaults to ~/.penguin/data (PENGUIN_HOME still overrides; the
installer keeps its binaries in ~/.penguin), and every Agent lives
under <project>/agents/<agent>/ — path helpers, the three
agent-enumeration scans, the system prompt, built-in Skills, tests
and docs all follow the new layout.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018ihk8iQuo3kv2aPjAYEPuR
This commit is contained in:
Yaowei Zheng
2026-07-19 14:06:53 +08:00
committed by GitHub
parent 056bed7aeb
commit 45bfae6e94
543 changed files with 92949 additions and 0 deletions
+643
View File
@@ -0,0 +1,643 @@
/**
* Agent and the `createAgent` entry point.
*
* `createAgent` is the unified way to create/load an Agent: it initializes Agent State
* if the directory is empty, otherwise loads by agentId.
* An Agent has exactly one Agent State and can run multiple times; the
* Workspace is determined when a Session is created.
*/
import fs from "node:fs/promises";
import path from "node:path";
import {
assertValidId,
assembleSystemPrompt,
buildToolConfig,
selectBuiltinToolsForModel,
DEFAULT_COMPACTION_PROMPT,
formatModelRef,
getModel,
listInstalledSkills,
loadAgentVault,
loadOrInitAgentState,
loadProjectConfig,
projectDir,
resolveModelRef,
scratchpadDir,
systemConfigPath,
tracesDir,
type AgentState,
type ModelRef,
type ProjectConfig,
} from "./state/index.js";
import { GenerativeModel, ToolCallIdAllocator } from "./llm/index.js";
import { Environment } from "./environment/index.js";
import {
Writer,
findLatestTraceFile,
latestSessionId as latestTraceSessionId,
readTraceTolerant,
resumeTrace,
} from "./trace/index.js";
import { Session } from "./session.js";
import {
createTempWorkspace,
formatSessionId,
sessionEnvironment,
} from "./internal/session-support.js";
import { userText, withOrigin } from "./omnimessage/index.js";
import type {
MessageOrigin,
OmniMessage,
TokenCounts,
ToolCallPayload,
} from "./omnimessage/index.js";
import { SUBAGENT_NAME } from "./environment/tools/run-subagent.js";
import { INPUT_SUBAGENT_NAME } from "./environment/tools/input-subagent.js";
import type { CompactionSettings } from "./engine/context-engine.js";
import type {
GenerativeModelConfig,
SubagentRunner,
ToolDefinition,
VisionDescriberService,
} from "./interfaces.js";
import type { ModelEntry } from "./state/index.js";
/**
* Maximum subagent spawn depth. Currently capped at 1 level (a subagent cannot spawn
* another subagent); the depth mechanism is designed to support multiple levels —
* raise this constant to allow deeper nesting.
*/
const MAX_SUBAGENT_DEPTH = 1;
export interface CreateAgentOptions {
agentId?: string;
projectId?: string;
/** Local data root directory; defaults to `resolveRoot()` (PENGUIN_HOME or ~/.penguin/data). */
root?: string;
}
export interface CreateSessionOptions {
/** Workspace for this run; if unspecified, a temporary Workspace is created under the Agent directory. */
workspaceDir?: string;
/** Model used for this Session (upstream model_id); if unspecified, uses the Project's default Model. */
modelId?: string;
/**
* Provider grouping for `modelId` (a paired reference); if omitted, resolved via
* `resolveModelRef` semantics — `model_id` only resolves if it is a globally unique
* exact match in the config; zero or multiple matches produce a clear error.
*/
provider?: string;
/** Explicit credentials; if unspecified, falls back to credentials in the Project config, then to AgentHub reading environment variables. */
apiKey?: string;
baseUrl?: string;
/** Internal use: this Session's depth in the subagent spawn chain (0 at the top level), used to cap spawn depth. */
subagentDepth?: number;
}
export interface ResumeSessionOptions {
/** Id of the Session to resume. */
sessionId: string;
/** Explicit credentials; if unspecified, falls back to credentials in the Project config, then to AgentHub reading environment variables. */
apiKey?: string;
baseUrl?: string;
}
/**
* Effective compaction threshold: capped at 75% of the model's `context_window` —
* the threshold must stay well below the hard window limit, otherwise small-window
* models get rejected by the provider (a non-retryable 400) before compaction even
* triggers, and the compaction request itself (old context + prompt + summary output)
* also needs headroom. Not clamped when `<=0` (disabled) or the window is unknown.
*/
export function effectiveMaxContextLength(configured: number, contextWindow: unknown): number {
if (configured <= 0) return configured;
if (typeof contextWindow !== "number") return configured;
return Math.min(configured, Math.floor(contextWindow * 0.75));
}
/** Create or load an Agent. */
export async function createAgent(opts: CreateAgentOptions = {}): Promise<Agent> {
const state = await loadOrInitAgentState(opts);
const projectConfig = await loadProjectConfig(state.root, state.projectId);
return new Agent(state, projectConfig);
}
export class Agent {
constructor(
readonly state: AgentState,
readonly projectConfig: ProjectConfig,
) {}
/**
* Create a Session in the specified (or a temporary) Workspace.
* Docs: /docs/sessions-and-traces § "Run model".
*/
async createSession(opts: CreateSessionOptions = {}): Promise<Session> {
// Model is validated first (before creating the Workspace, so failure leaves no
// temp directory behind): the reference must resolve to an entry in the Project
// config (the (provider, model_id) pair is the unique key); a reference
// outside the config throws immediately rather than passing silently — otherwise
// credentials, pricing, and the context window would all be unavailable.
if (opts.modelId === undefined && opts.provider !== undefined) {
throw new Error(
"指定了 provider 却未指定 modelId:模型引用须成对给出(provider 不能单独使用)。",
);
}
let ref: ModelRef;
if (opts.modelId !== undefined) {
// The only entry point for resolving an "omitted provider" reference (resolveModelRef): three branches — unique match / zero matches / ambiguous.
ref = resolveModelRef(this.projectConfig, opts.modelId, opts.provider);
} else if (this.projectConfig.default_model) {
ref = this.projectConfig.default_model;
} else {
throw new Error(
"未指定 modelId,且 Project 配置中没有 default_model。请用 `penguin config model add/default` 设置默认模型。",
);
}
const modelEntry = getModel(this.projectConfig, ref);
if (!modelEntry) {
throw new Error(
`Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`,
);
}
// Credentials are inlined on the model entry (single config file); an
// explicit argument takes priority, falling back to AgentHub reading env vars
// when both are absent.
const apiKey = opts.apiKey ?? modelEntry.api_key;
const baseUrl = opts.baseUrl ?? modelEntry.base_url;
// An explicit Workspace must already exist as a directory: if it
// doesn't, throw rather than auto-create (to avoid a typo silently working in
// the wrong location); a temp Workspace is only created when unspecified.
let workspaceDir: string;
if (opts.workspaceDir) {
workspaceDir = path.resolve(opts.workspaceDir);
let stat;
try {
stat = await fs.stat(workspaceDir);
} catch {
throw new Error(
`Workspace 不存在:${workspaceDir}。请指定一个已存在的目录,或不指定 Workspace 以使用临时目录。`,
);
}
if (!stat.isDirectory()) {
throw new Error(`Workspace 不是目录:${workspaceDir}。`);
}
} else {
workspaceDir = await createTempWorkspace(
this.state.root,
this.state.projectId,
this.state.agentId,
);
}
const sessionId = formatSessionId();
const subagentDepth = opts.subagentDepth ?? 0;
// Agent-level vault (agent_state/.vault.toml) and installed Skills: read the current values each time a Session is created.
const vault = await loadAgentVault(this.state.root, this.state.projectId, this.state.agentId);
const installedSkills = await listInstalledSkills(
this.state.root,
this.state.projectId,
this.state.agentId,
);
// The assembled system prompt goes both to the LLM and into session_meta (so the
// Trace can audit the actual effective value). The vault only injects **key names**
// into the prompt (so the model knows which API keys are available); values only
// go into the subprocess environment. Skills only inject metadata (name and
// description); the model reads the body on demand via shell.
const systemPrompt = assembleSystemPrompt(
this.state,
sessionEnvironment(workspaceDir, sessionId, {
agentId: this.state.agentId,
projectDir: projectDir(this.state.root, this.state.projectId),
}),
Object.keys(vault),
installedSkills,
);
const rt = await this.buildRuntime({
workspaceDir,
modelEntry,
apiKey,
baseUrl,
systemPrompt,
subagentDepth,
vault,
});
const trace = new Writer({
tracesDir: tracesDir(this.state.root, this.state.projectId, this.state.agentId),
sessionId,
});
return new Session({
meta: {
session_id: sessionId,
provider: modelEntry.provider,
model_id: modelEntry.model_id,
model_context_window: modelEntry.context_window ?? "unknown",
system_prompt: systemPrompt,
tools: rt.tools,
thinking_level: this.state.systemConfig.model?.thinking_level ?? "default",
agent_state: this.state.stateDir,
workspace: workspaceDir,
},
llm: rt.llm,
environment: rt.environment,
trace,
createLLM: rt.createLLM,
createBareLLM: rt.createBareLLM,
compaction: rt.compaction,
// Model doesn't support images: input images are written to the session scratchpad and their paths appended to the text (viewed via describe_image).
...(modelEntry.vision === false
? {
inputImagesDir: path.join(
scratchpadDir(this.state.root, this.state.projectId, this.state.agentId),
sessionId,
),
}
: {}),
// Max turns comes from the Agent's system_config (runtime parameters belong to the Agent config).
...(this.state.systemConfig.max_turns !== undefined
? { maxTurns: this.state.systemConfig.max_turns }
: {}),
});
}
/**
* Resume an existing Session and continue the conversation.
*
* The resume source is the Session's **latest-index** Trace file: runtime config is
* read from its `session_meta` (Model, the original system prompt text, and the
* Workspace all carry over from the original Session and cannot be changed), while
* tools and Environment are reassembled from the current Agent State. The replayed,
* already-committed history is injected once via AgentHub's setHistory (used only on
* resume); any leftover input is rebuilt as carry-over (paired fallback placeholders
* are synthesized in memory only, never written to the Trace). Messages after resume
* continue in the original Trace file (the file follows the context, not the date),
* and Token / turn-count stats carry over from their original values.
* Docs: /docs/sessions-and-traces § "Session recovery".
*/
async resumeSession(opts: ResumeSessionOptions): Promise<Session> {
const { sessionId } = opts;
const dir = tracesDir(this.state.root, this.state.projectId, this.state.agentId);
const located = await findLatestTraceFile(dir, sessionId);
if (!located) {
throw new Error(`Session 不存在:${sessionId}(在 ${dir} 下未找到对应 Trace 文件)。`);
}
const resumed = resumeTrace(await readTraceTolerant(located.path));
if (!resumed.meta) {
throw new Error(`Trace 缺少 session_meta,无法恢复:${located.path}`);
}
const meta = resumed.meta.payload;
// Model reference is stored as a pair in session_meta; a missing provider means legacy data (no migration since the product hasn't shipped yet).
if (typeof meta.provider !== "string") {
throw new Error(
`Trace 来自旧版本数据(session_meta 缺少 provider,模型引用未分列):${located.path}。请删除数据目录后重建会话。`,
);
}
// The Workspace carries over from the original Session and must still exist (throw if missing, never auto-create).
const workspaceDir = meta.workspace;
let stat;
try {
stat = await fs.stat(workspaceDir);
} catch {
throw new Error(`原 Session 的 Workspace 已不存在:${workspaceDir},无法恢复。`);
}
if (!stat.isDirectory()) {
throw new Error(`原 Session 的 Workspace 不是目录:${workspaceDir},无法恢复。`);
}
// The Model carries over from the original Session (paired reference) and must still be present in the Project config.
const ref: ModelRef = { provider: meta.provider, model_id: meta.model_id };
const modelEntry = getModel(this.projectConfig, ref);
if (!modelEntry) {
throw new Error(
`原 Session 的 Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model add\` 重新配置后再恢复。`,
);
}
const apiKey = opts.apiKey ?? modelEntry.api_key;
const baseUrl = opts.baseUrl ?? modelEntry.base_url;
// Tools and Environment are reassembled from the current Agent State (tool
// definitions are passed with every Request and aren't part of the history); the
// system prompt uses the original text recorded in the Trace (identical to the
// original history); the vault uses current values (it's injected into the
// subprocess environment, not the history, so a resumed Session should get the
// latest keys too).
const rt = await this.buildRuntime({
workspaceDir,
modelEntry,
apiKey,
baseUrl,
systemPrompt: meta.system_prompt,
subagentDepth: 0,
vault: await loadAgentVault(this.state.root, this.state.projectId, this.state.agentId),
});
// History is injected once into a fresh context object (setHistory is only used
// on resume); Session cumulative Token counts carry over. Wrap the error
// descriptively: bad tool arguments in the history (e.g. truncated JSON written by
// a third-party OpenAI-compatible endpoint) throw a raw SyntaxError during
// conversion, so the error must indicate Trace history corruption rather than a
// regular runtime error.
if (resumed.history.length > 0) {
try {
rt.llm.setHistory(resumed.history);
} catch (err) {
const detail = err instanceof Error ? err.message : String(err);
throw new Error(
`恢复失败:Trace 历史无法注入(记录可能损坏,如非法的工具参数 JSON):${detail}`,
);
}
}
rt.llm.sessionTokens = resumed.sessionTokens;
// Continue writing to the original Trace file (the Trace only records real messages; synthesized paired placeholders are re-emitted in memory alongside carry-over).
const trace = new Writer({
tracesDir: dir,
sessionId,
dateDir: located.dateDir,
startIndex: located.index,
});
return new Session({
meta: {
session_id: sessionId,
provider: modelEntry.provider,
model_id: modelEntry.model_id,
model_context_window: modelEntry.context_window ?? "unknown",
system_prompt: meta.system_prompt,
tools: rt.tools,
thinking_level: this.state.systemConfig.model?.thinking_level ?? "default",
agent_state: this.state.stateDir,
workspace: workspaceDir,
},
llm: rt.llm,
environment: rt.environment,
trace,
createLLM: rt.createLLM,
createBareLLM: rt.createBareLLM,
compaction: rt.compaction,
// Model doesn't support images: input images are written to the session scratchpad and their paths appended to the text (viewed via describe_image).
...(modelEntry.vision === false
? {
inputImagesDir: path.join(
scratchpadDir(this.state.root, this.state.projectId, this.state.agentId),
sessionId,
),
}
: {}),
...(this.state.systemConfig.max_turns !== undefined
? { maxTurns: this.state.systemConfig.max_turns }
: {}),
// session_meta is already in the original Trace file, so it isn't rewritten; on the first write after a compaction-triggered rotation, the file is split first.
metaAlreadyWritten: true,
initialEngineState: {
carryOver: resumed.carryOver,
...(resumed.pendingSummary ? { pendingSummary: resumed.pendingSummary } : {}),
sessionTurns: resumed.sessionTurns,
sessionTokens: resumed.sessionTokens,
lastRequestTotal: resumed.lastRequestTotal,
pendingTraceRotation: resumed.contextClosed,
},
resumedHistory: resumed.renderMessages,
});
}
/** Id of the most recent Session under the current Agent (determined by the timestamp in session_id); returns null if there is no Session. */
async latestSessionId(): Promise<string | null> {
return latestTraceSessionId(
tracesDir(this.state.root, this.state.projectId, this.state.agentId),
);
}
/**
* Assemble a Session's runtime components (shared by createSession and
* resumeSession): the child-Agent runner, Environment and tools, the LLM object
* and its post-compaction rebuild factory, and the compaction config.
*/
private async buildRuntime(args: {
workspaceDir: string;
/** This Session's Model entry: the caller (createSession / resumeSession) has already validated it exists in the config. */
modelEntry: ModelEntry;
apiKey: string | undefined;
baseUrl: string | undefined;
systemPrompt: string;
subagentDepth: number;
vault: Record<string, string>;
}): Promise<{
environment: Environment;
tools: ToolDefinition[];
llm: GenerativeModel;
createLLM: (sessionTokens: TokenCounts) => GenerativeModel;
createBareLLM: () => GenerativeModel;
compaction: CompactionSettings;
}> {
const { workspaceDir, modelEntry, apiKey, baseUrl, systemPrompt, subagentDepth, vault } = args;
// Child-Agent runner: injected into the run_subagent tool so it doesn't need to
// depend on Agent/Session (breaking a circular dependency). The model can
// optionally choose agentId (omitted = call the current Agent) and modelId
// (omitted = Project default). Precheck errors (depth limit exceeded / agent
// doesn't exist) are expressed as throws, which the Environment collapses to failed.
// Docs: /docs/interfaces § "Subagent interfaces"
const parentAgent = this;
const { root, projectId, agentId: parentAgentId } = this.state;
const subagentRunner: SubagentRunner = {
// Spawn and run are separate: the same child Session can run for multiple turns
// (continuing via input_subagent appending a prompt); resource cleanup is
// consolidated in handle.dispose (called by the managing ManagedSubagentSession).
async spawn({ agentId, modelId }) {
if (subagentDepth >= MAX_SUBAGENT_DEPTH) {
throw new Error(
`subagent depth limit ${MAX_SUBAGENT_DEPTH} reached; not spawning another subagent`,
);
}
if (agentId !== undefined && agentId !== parentAgentId) {
try {
assertValidId("agent_id", agentId);
await fs.access(systemConfigPath(root, projectId, agentId));
} catch {
throw new Error(
`subagent error: agent "${agentId}" does not exist or is not accessible`,
);
}
}
const childAgent =
agentId !== undefined && agentId !== parentAgentId
? await createAgent({ root, projectId, agentId })
: parentAgent;
const childSession = await childAgent.createSession({
workspaceDir,
...(modelId !== undefined ? { modelId } : {}),
subagentDepth: subagentDepth + 1,
});
// All child-session messages are tagged with an origin (the child Session id,
// prepended as one hop from outer to inner); the first turn forwards the
// child's session_meta first (including agent_state and other metadata) so the
// parent frontend can recognize the nested session (for rendering, stats,
// approval visibility); the parent Trace skips these accordingly (the child
// Session has its own Trace, linked by session id).
const hop: MessageOrigin = childSession.sessionId;
let metaSent = false;
return {
sessionId: hop,
async *run({ prompt, signal, approve }) {
if (!metaSent) {
metaSent = true;
yield withOrigin(childSession.metaMessage, hop);
}
// Pass through the parent's approval callback: the child Session inherits
// the parent Agent's approval mode (with no callback, the child engine
// defaults to deny). The tool_call received for approval also carries the
// origin, so the approval UI can identify which tool a subagent is calling.
const childApprove = approve
? (tc: OmniMessage<ToolCallPayload>) => approve(withOrigin(tc, hop))
: undefined;
for await (const msg of childSession.run([userText(prompt)], {
...(signal ? { signal } : {}),
...(childApprove ? { approve: childApprove } : {}),
})) {
yield withOrigin(msg, hop);
}
},
dispose() {
childSession.dispose();
},
};
},
};
// Tool exposure is capped by depth: a (leaf) child Agent that has reached the
// max spawn depth no longer gets run_subagent or input_subagent (the latter
// depends on the subagent_id produced by the former, so exposing it alone is
// meaningless).
const canSpawn = subagentDepth < MAX_SUBAGENT_DEPTH;
const baseToolConfig = buildToolConfig(this.state);
// Select tool entries by the session model's type (marked via forModel: vision
// models use read_image, text-only models use describe_image; entries without
// this marker are unaffected).
const modelVision = modelEntry.vision !== false;
let customTools = selectBuiltinToolsForModel(baseToolConfig.customTools, modelVision);
if (!canSpawn) {
customTools = customTools.filter(
(d) => d.name !== SUBAGENT_NAME && d.name !== INPUT_SUBAGENT_NAME,
);
}
const toolConfig = { ...baseToolConfig, customTools };
// When the session model doesn't support images (vision=false): inject a vision
// model service for describe_image (forModel: "text-only", selected by the filter
// above) — images are described by the Project config's vision_model (a paired
// reference), and the tool returns text. Even when unconfigured or invalid, it is
// still injected (modelId=null); the tool then finishes with a failed explanation,
// and images are never allowed into that session's history.
let visionDescriber: VisionDescriberService | undefined;
if (modelEntry.vision === false) {
const visionRef = this.projectConfig.vision_model;
const visionEntry = visionRef ? getModel(this.projectConfig, visionRef) : undefined;
if (visionEntry && visionEntry.vision !== false) {
visionDescriber = {
// The model attribution in the tool output matches the request's source: both are the entry's upstream model_id.
modelId: visionEntry.model_id,
createLLM: () =>
new GenerativeModel({
modelId: visionEntry.model_id,
...(visionEntry.api_key !== undefined ? { apiKey: visionEntry.api_key } : {}),
...(visionEntry.base_url !== undefined ? { baseUrl: visionEntry.base_url } : {}),
...(visionEntry.client_type !== undefined
? { clientType: visionEntry.client_type }
: {}),
tools: [],
thinkingLevel: "none",
maxTokens: 2048,
requestTimeoutMs: 60_000,
}),
};
} else {
visionDescriber = { modelId: null };
}
}
// Environment binds the Workspace and tool config; tools are listed first so
// GenerativeModel can be initialized. Vault environment variables are injected
// into command subprocesses (shared by createSession and resumeSession; the
// caller reads the current agent_state/.vault.toml); a child Agent loads **its
// own** vault via createAgent rather than inheriting the parent's.
const environment = new Environment({
workspaceDir,
toolConfig,
services: { subagentRunner, ...(visionDescriber ? { visionDescriber } : {}) },
...(Object.keys(vault).length > 0 ? { vault } : {}),
});
const tools = await environment.listTools();
// LLM constructor args are extracted into a constant so they can be reused as-is when
// rebuilding a new LLM object after compaction (with a fresh model context) — the system
// prompt and tool definitions aren't part of the compacted history, so the new object keeps
// them unchanged. The model id sent to AgentHub is always the entry's upstream `model_id`
// (client_type inference/passing follows it); session_meta, Trace, usage, pricing, and catalog
// matching all use the (provider, model_id) pair as the primary key.
// The tool_call_id uniqueness registry is shared with the new LLM rebuilt from llmConfig after
// compaction: its uniqueness scope is the Session's whole render span, so same-named tool calls
// after compaction don't collide with earlier tool cards' ids.
const llmConfig: GenerativeModelConfig = {
modelId: modelEntry.model_id,
toolCallIds: new ToolCallIdAllocator(),
...(apiKey !== undefined ? { apiKey } : {}),
...(baseUrl !== undefined ? { baseUrl } : {}),
...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}),
tools,
systemPrompt,
...(modelEntry.context_window !== undefined
? { contextWindow: modelEntry.context_window }
: {}),
...(this.state.systemConfig.model?.max_tokens !== undefined
? { maxTokens: this.state.systemConfig.model.max_tokens }
: {}),
...(this.state.systemConfig.model?.thinking_level !== undefined
? { thinkingLevel: this.state.systemConfig.model.thinking_level }
: {}),
...(this.state.systemConfig.model?.timeoutMs !== undefined
? { requestTimeoutMs: this.state.systemConfig.model.timeoutMs }
: {}),
};
const llm = new GenerativeModel(llmConfig);
const createLLM = (sessionTokens: TokenCounts): GenerativeModel => {
const next = new GenerativeModel(llmConfig);
// Carries over the Session's cumulative Token counts, so token_usage.session stays continuous across compaction.
next.sessionTokens = sessionTokens;
return next;
};
// Bare LLM for one-off out-of-band requests (meta requests like generateTitle):
// same Model/credentials, no tools, no system prompt, thinking disabled, a small
// output cap, and an independent timeout.
const createBareLLM = (): GenerativeModel =>
new GenerativeModel({
modelId: modelEntry.model_id,
...(apiKey !== undefined ? { apiKey } : {}),
...(baseUrl !== undefined ? { baseUrl } : {}),
...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}),
tools: [],
thinkingLevel: "none",
maxTokens: 300,
requestTimeoutMs: 30_000,
});
// Compaction config: defaults are filled in here; an unknown mode falls back to summarize (the default).
const compactionConfig = this.state.systemConfig.compaction;
const compaction: CompactionSettings = {
maxContextLength: effectiveMaxContextLength(
compactionConfig?.max_context_length ?? 128000,
modelEntry.context_window,
),
maxSessionTurns: compactionConfig?.max_session_turns ?? -1,
mode: compactionConfig?.mode === "discard" ? "discard" : "summarize",
prompt: compactionConfig?.prompt ?? DEFAULT_COMPACTION_PROMPT,
};
return { environment, tools, llm, createLLM, createBareLLM, compaction };
}
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,380 @@
/**
* Environment —— executes approved tool calls inside the Workspace.
*
* Environment has no knowledge of any specific tool: it only assembles the tool names supported
* by ToolConfig into BuiltinTool instances (see `environment/tools/`), and dispatches execution
* by looking up the tool name. Adding a new built-in tool only requires implementing BuiltinTool
* and registering it — no changes to this file needed. Tool call **rendering** is not core's
* concern; it's handled by the CLI / Web frontend.
*
* The **framing and finalization** of the tool stream is handled uniformly by Environment:
* - Entering execution immediately emits `start`; the tool only needs to yield output deltas
* (its own start/stop are ignored);
* - Output is truncated online **front-to-back** by maxOutputLength (head is kept, forwarding
* stops once exceeded); the truncation marker, the tool's self-reported end marker
* (`ToolResult.note`, e.g. exit code — appended outside the truncation, never lost even when
* long output is truncated), and timeout/interruption/error markers are all emitted as part
* of the stream — **the content produced by concatenating streamed chunks matches the full
* message exactly**;
* - Nested session messages carrying an origin marker (e.g. forwarded from run_subagent) pass
* through unchanged, taking no part in this tool's output or finalization;
* - Argument parsing failures, unknown tool names, tool throws, and other exceptions all
* collapse into an explanatory, complete `tool_call_output` — never throws — and **output is
* never empty under any circumstance**.
* Docs: /docs/tools § "Execution contract".
*/
import { partialToolCallOutput, toolCallOutput } from "../omnimessage/index.js";
import type { OmniMessage, StopReason } from "../omnimessage/index.js";
import type {
EnvironmentConfig,
EnvironmentInterface,
ToolConfig,
ToolDefinition,
ToolExecutionRequest,
ToolPermission,
} from "../interfaces.js";
import type { BuiltinTool, ToolResult } from "./tools/types.js";
import { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js";
import { CommandSessionManager } from "./tools/command/index.js";
import { SubagentSessionManager } from "./tools/subagent/index.js";
/** Default cap on tool output truncation (characters). */
const DEFAULT_MAX_OUTPUT_LENGTH = 16000;
/** Default timeout cap for a single tool call (milliseconds); <=0 disables it (every tool must be bound by timeoutMs). */
const DEFAULT_TOOL_TIMEOUT_MS = 120000;
/** Marker appended to the result when a tool is interrupted by the user. */
const TOOL_ABORTED_NOTE = "[interrupted: tool aborted by user]";
/** Placeholder marker used when a tool produces no output at all (tool_call_output content is never empty). */
const TOOL_EMPTY_NOTE = "[no output]";
/**
* Explanation for a failed argument JSON parse. The normal pipeline never reaches this: bad
* JSON already throws during AgentHub's parsing stage, and the LLM layer finalizes it as
* malformed for the engine to reconnect (see generative-model.ts) — it's never dispatched into
* Environment as a completed tool_call. This function is only a defensive fallback for the
* public interface.
*/
function describeArgumentsError(name: string, raw: string, err: unknown): string {
const detail = err instanceof Error ? err.message : String(err);
if (raw.trim() === "") {
return `Tool call "${name}" failed: the arguments field is empty. Re-issue the call with a complete JSON object.`;
}
return `Tool call "${name}" failed: the arguments are not valid JSON (${detail}). Re-issue the call with one complete, valid JSON object.`;
}
/** Appends a marker after existing content: newline-joins if content is non-empty, otherwise just returns the marker. */
function appendNote(base: string, note: string): string {
return base ? `${base}\n${note}` : note;
}
/** The delta needed to stream out `note` on top of existing content `base` (includes separator, same basis as appendNote). */
function noteSuffix(base: string, note: string): string {
return base ? `\n${note}` : note;
}
export class Environment implements EnvironmentInterface {
private readonly workspaceDir: string;
private readonly toolConfig: ToolConfig;
/** Assembled built-in tools: tool name -> BuiltinTool. Only tools supported by the registry and present in config. */
private readonly tools: Map<string, BuiltinTool>;
/** Long-running command session registry: constructed within this Environment and shared between exec_command / input_command. */
private readonly commandSessions: CommandSessionManager;
/** Background subagent session registry: constructed within this Environment and shared between run_subagent / input_subagent. */
private readonly subagentSessions: SubagentSessionManager;
constructor(config: EnvironmentConfig) {
this.workspaceDir = config.workspaceDir;
this.toolConfig = config.toolConfig;
this.tools = new Map();
// The background session registry is created alongside Environment (one per Session) and
// injected into whichever tools need it; all sessions are finalized together on dispose.
// The vault environment variables are injected into child processes by the command session
// registry at spawn time.
this.commandSessions = new CommandSessionManager(
config.vault !== undefined ? { vault: config.vault } : {},
);
this.subagentSessions = new SubagentSessionManager();
const services = {
...config.services,
commandSessions: this.commandSessions,
subagentSessions: this.subagentSessions,
};
// Assemble the tools supported by config into BuiltinTool instances; unrecognized tool
// names are skipped (neither exposed to the LLM nor executable).
for (const def of config.toolConfig.customTools) {
const factory = BUILTIN_TOOL_FACTORIES[def.name];
if (factory) this.tools.set(def.name, factory(def, services));
}
}
/** Releases runtime resources held by Environment: finalizes all managed background sessions (command and subagent). Idempotent. */
dispose(): void {
this.commandSessions.dispose();
this.subagentSessions.dispose();
}
/**
* Lists tools available to the current Session, for context_engine to initialize GenerativeModel.
* Only lists tools that have been assembled (i.e. supported by the registry) — tool names
* unrecognized in config are not exposed to the LLM (consistent with the constructor);
* the definition (description/parameters) treats **the config entry as the single source of
* truth** — factories must not rewrite the definition at runtime; where a differentiated
* implementation is needed, use a separate explicit tool-name entry with a `forModel`
* annotation (e.g. read_image / describe_image).
* Only exposes `{name, description, parameters}`, dropping permission/maxOutputLength.
* MCP Server config flows into Environment via toolConfig; enumerating concrete MCP tools
* is left to a later adapter layer.
*/
async listTools(): Promise<ToolDefinition[]> {
return this.toolConfig.customTools
.filter((tool) => this.tools.has(tool.name))
.map((tool) => ({
name: tool.name,
description: tool.description,
...(tool.parameters !== undefined ? { parameters: tool.parameters } : {}),
}));
}
/** Looks up a tool's permission level (for the frontend's permission-mode decisions); returns undefined for an unknown tool. */
toolPermission(name: string): ToolPermission | undefined {
return this.toolConfig.customTools.find((t) => t.name === name)?.permission;
}
/**
* Executes an approved tool call, streaming `partial_tool_call_output` and a final
* `tool_call_output`; nested messages carrying origin pass through unchanged. Dispatches by
* looking up the tool name; any exception collapses into an explanatory output — never throws.
*
* The priority for deciding stop_reason is: user interruption > timeout > tool throw > tool
* self-report. Interruption is determined by the `signal` held by Environment, and is
* compatible with both a tool self-reporting aborted and an AbortError raised by the
* interruption. An internal abort raised by a timeout does not count as a user interruption —
* it's finalized as failed, with the timeout reason written into the output.
* Docs: /docs/tools § "Execution contract".
*/
async *executeTool(request: ToolExecutionRequest): AsyncGenerator<OmniMessage> {
const payload = request.toolCall.payload;
// tool_call_id is passed through unchanged, so context_engine and the LLM can associate the
// request with its result.
const toolCallId = payload.tool_call_id;
const name = payload.name;
// Every path is framed uniformly by Environment: entering execution emits start; the end
// uniformly emits stop + the full message.
yield partialToolCallOutput({ eventType: "start", toolCallId });
const tool = this.tools.get(name);
if (!tool) {
yield* emitFailure(toolCallId, `Unknown tool: ${name}`);
return;
}
// Parse the tool's argument JSON; a parse failure collapses into an explanatory output
// (also streamed, so the frontend can render it).
let parsed: unknown;
try {
parsed = JSON.parse(payload.arguments);
} catch (err) {
yield* emitFailure(toolCallId, describeArgumentsError(name, payload.arguments, err));
return;
}
const args =
parsed !== null && typeof parsed === "object" ? (parsed as Record<string, unknown>) : {};
const maxOutputLength = tool.definition.maxOutputLength ?? DEFAULT_MAX_OUTPUT_LENGTH;
const timeoutMs = tool.definition.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS;
const signal = request.signal;
// User interruption and tool timeout are merged into a single internal signal handed to the
// tool: either one triggers abortion of execution.
// The timeout constraint is enforced uniformly by Environment for all tools; the
// tool only needs to respond to signal.
const ac = new AbortController();
if (signal?.aborted) ac.abort();
const onAbort = (): void => ac.abort();
signal?.addEventListener("abort", onAbort, { once: true });
let timedOut = false;
const timer =
timeoutMs > 0
? setTimeout(() => {
timedOut = true;
ac.abort();
}, timeoutMs)
: null;
timer?.unref?.();
// Consume the tool stream: content deltas are forwarded after online front-truncation;
// nested messages pass through; manual iteration to capture the generator's return value.
let streamed = ""; // Content forwarded so far (<= maxOutputLength)
let contentLen = 0; // Total length of content produced by the tool (including truncated/discarded parts)
let toolOutput: string | null = null; // Fallback: content basis when the tool produces a full message itself
let selfReported: StopReason | undefined; // Tool's self-reported stop reason (return value takes priority over the full message)
let selfNote: string | null = null; // Tool's self-reported end marker (e.g. exit code), appended outside truncation
let selfImages: string[] | undefined; // Tool's self-reported images (data URL), carried via a single streamed delta and the full message
let thrown: unknown = null;
const gen = tool.execute(args, {
workspaceDir: this.workspaceDir,
toolCallId,
signal: ac.signal,
// Pass through the parent's approve callback (run_subagent uses it so the child Session
// inherits the parent's approval mode; other tools ignore it).
...(request.approve ? { approve: request.approve } : {}),
});
try {
for (;;) {
const res = await gen.next();
if (res.done) {
const result: ToolResult | void = res.value;
if (result?.stopReason) selfReported = result.stopReason;
if (result?.note) selfNote = result.note;
if (result?.images && result.images.length > 0) selfImages = result.images;
break;
}
const out = res.value;
if (out.origin && out.origin.length > 0) {
yield out; // Nested session message: pass through unchanged, not part of this tool's output/finalization
continue;
}
const p = out.payload as {
type?: string;
event_type?: string;
stop_reason?: string;
output?: string;
};
if (p.type === "partial_tool_call_output") {
// Only takes delta content; start/stop are ignored (framing is uniformly handled by Environment).
if (p.event_type !== "delta" || !p.output) continue;
contentLen += p.output.length;
// maxOutputLength <= 0 means truncation is disabled (same semantics as timeoutMs).
const room =
maxOutputLength > 0 ? maxOutputLength - streamed.length : Number.POSITIVE_INFINITY;
if (room > 0) {
const chunk = p.output.length > room ? p.output.slice(0, room) : p.output;
streamed += chunk;
// Rebuild the delta: tool_call_id is uniformly enforced by Environment, never trusting the tool's own value.
yield partialToolCallOutput({
eventType: "delta",
output: chunk,
toolCallId,
});
}
} else if (p.type === "tool_call_output") {
// Fallback: if the tool still produces a full message, use it as the basis for content and stop reason (not needed under the new contract).
toolOutput = p.output ?? "";
if (selfReported === undefined && p.stop_reason) {
selfReported = p.stop_reason as StopReason;
}
} else {
// Other message types without origin: protocol misuse, ignore and warn (keep the parent stream clean).
process.stderr.write(
`[penguin] tool "${name}" yielded unexpected message type "${p.type}"; ignored.\n`,
);
}
}
} catch (err) {
// A tool throw also collapses into the uniform finalization: keep already-streamed content, don't discard produced output.
thrown = err;
} finally {
if (timer) clearTimeout(timer);
signal?.removeEventListener("abort", onAbort);
}
// Uniform finalization. Content basis = the tool's self-produced full message (fallback
// path) or the already-forwarded delta; after front-truncating to the cap, the truncation
// marker and interruption/timeout/error markers are appended in turn, all made up via
// streamed deltas — streamed concatenation == the full message.
const contentBase = toolOutput ?? streamed;
const capped =
maxOutputLength > 0 && contentBase.length > maxOutputLength
? contentBase.slice(0, maxOutputLength)
: contentBase;
const truncated = capped.length < contentBase.length || contentLen > streamed.length;
const aborted =
signal?.aborted === true ||
(!timedOut &&
(selfReported === "aborted" ||
(thrown as { name?: string } | null)?.name === "AbortError"));
let stopReason: StopReason;
const notes: string[] = [];
if (truncated) {
notes.push(`[output truncated: exceeded ${maxOutputLength} chars]`);
}
// The tool's self-reported end marker (e.g. exit code): appended outside the truncation —
// if treated as a content delta it would get cut off once long output hits the cap, and the
// model would misread a command that failed after printing lots of output as successful.
if (selfNote) {
notes.push(selfNote);
}
if (aborted) {
stopReason = "aborted";
notes.push(TOOL_ABORTED_NOTE);
} else if (timedOut) {
stopReason = "failed";
notes.push(`[tool timeout: exceeded ${timeoutMs}ms]`);
} else if (thrown != null) {
stopReason = "failed";
notes.push(`[tool error] ${thrown instanceof Error ? thrown.message : String(thrown)}`);
} else {
stopReason = selfReported ?? "completed";
}
// The tool's reply must never be empty: an empty tool_result leaves the model unable to
// tell "silent success" apart from "call failed", and some Providers outright reject empty
// content blocks.
if (capped === "" && notes.length === 0) {
notes.push(TOOL_EMPTY_NOTE);
}
const noteText = notes.join("\n");
const fullOutput = noteText ? appendNote(capped, noteText) : capped;
// Compensating content delta: if nothing was streamed, emit the whole thing at once; on the
// fallback path, emit the portion of the full message beyond the already-streamed prefix
// (if the tool is internally inconsistent, the full message wins — no further reconciliation).
let compensation = "";
if (streamed === "") compensation = capped;
else if (toolOutput !== null && capped.startsWith(streamed)) {
compensation = capped.slice(streamed.length);
}
if (compensation) {
yield partialToolCallOutput({
eventType: "delta",
output: compensation,
toolCallId,
});
}
if (noteText) {
yield partialToolCallOutput({
eventType: "delta",
output: noteSuffix(capped, noteText),
toolCallId,
});
}
// Images are made up via streaming: images are not delta'd — a single delta carries them
// all at once right before stop, and the full message carries them again — satisfying
// "streamed concatenation == full message" the same way text does (truncation only applies
// to text, never touches images).
// Only carried on normal completion; interruption/timeout/error paths carry no images, to keep finalization simple.
const images = stopReason === "completed" ? selfImages : undefined;
if (images) {
yield partialToolCallOutput({ eventType: "delta", toolCallId, images });
}
yield partialToolCallOutput({ eventType: "stop", toolCallId, stopReason });
yield toolCallOutput({
output: fullOutput,
toolCallId,
stopReason,
...(images ? { images } : {}),
});
}
}
/** Upfront failure (unknown tool/argument parse failure): delta(explanation) -> stop -> full failed output (start already emitted by the caller). */
function* emitFailure(toolCallId: string, message: string): Generator<OmniMessage> {
yield partialToolCallOutput({ eventType: "delta", output: message, toolCallId });
yield partialToolCallOutput({ eventType: "stop", toolCallId, stopReason: "failed" });
yield toolCallOutput({ output: message, toolCallId, stopReason: "failed" });
}
+14
View File
@@ -0,0 +1,14 @@
/**
* Environment module barrel — exports the environment interface implementation and builtin tool abstractions.
*/
export { Environment } from "./environment.js";
export type { BuiltinTool, ToolExecutionContext } from "./tools/types.js";
export { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js";
export type { BuiltinToolFactory } from "./tools/registry.js";
export { createExecCommandTool, EXEC_COMMAND_NAME } from "./tools/exec-command.js";
export { createInputCommandTool, INPUT_COMMAND_NAME } from "./tools/input-command.js";
export { createSubagentTool, SUBAGENT_NAME } from "./tools/run-subagent.js";
export { createInputSubagentTool, INPUT_SUBAGENT_NAME } from "./tools/input-subagent.js";
export { CommandSessionManager, ManagedSession } from "./tools/command/index.js";
export type { ProcessExit, SpawnOptions } from "./tools/command/index.js";
export { SubagentSessionManager, ManagedSubagentSession } from "./tools/subagent/index.js";
@@ -0,0 +1,44 @@
/**
* CappedTextBuffer — capacity-capped unread-text buffer shared by background sessions.
*
* When over capacity, drops the oldest content (keeping the tail) and tallies the dropped count;
* `drain()` prefixes a marker noting the drop count when taking all unread content (guards
* against a chatty background process / sub-agent blowing up memory, see each session class's
* capacity constant).
*/
export class CappedTextBuffer {
private text = "";
private omitted = 0; // Characters dropped due to the capacity cap, not yet read
/** `dropLabel` is used in the drop-marker text, e.g. "earlier output" / "earlier subagent output". */
constructor(
private readonly cap: number,
private readonly dropLabel: string,
) {}
get isEmpty(): boolean {
return this.text.length === 0 && this.omitted === 0;
}
append(chunk: string): void {
this.text += chunk;
if (this.text.length > this.cap) {
const drop = this.text.length - this.cap;
this.text = this.text.slice(drop); // Keep the newest (tail), drop the oldest
this.omitted += drop;
}
}
/** Takes the current unread content (including the drop marker); clears the buffer. */
drain(): string {
if (this.isEmpty) return "";
const b = this.text;
this.text = "";
if (this.omitted > 0) {
const n = this.omitted;
this.omitted = 0;
return `[... ${n} chars of ${this.dropLabel} dropped ...]\n${b}`;
}
return b;
}
}
@@ -0,0 +1,8 @@
/**
* Barrel for shared background-session infrastructure.
*/
export { BackgroundRegistry } from "./registry.js";
export type { BackgroundTask } from "./registry.js";
export { clampYield } from "./limits.js";
export { WakeSignal } from "./wake-signal.js";
export { CappedTextBuffer } from "./capped-buffer.js";
@@ -0,0 +1,23 @@
/**
* Yield-time clamping shared by background-session tools.
*
* `yield_time_ms` is a soft budget for a single tool call: "wait at most until the session ends
* or this duration expires" (expiry yields, it's not a failure). Only a lower bound is set: a
* wait that's too short isn't meaningful and just adds round trips. The upper bound is no longer
* an independent constant — it's derived from the tool's own `timeoutMs` (with a reserved
* margin, so the yield happens before the Environment's timeout fallback fires); no upper bound
* is set when `timeoutMs <= 0` (disabled).
*/
/** Lower bound for yield time (ms). */
export const MIN_YIELD_MS = 250;
/** Margin (ms) reserved between the yield upper bound and the tool's `timeoutMs`: the yield must happen before the timeout fallback. */
const TIMEOUT_MARGIN_MS = 1_000;
/** Clamps the raw argument to `[MIN_YIELD_MS, timeoutMs - margin]`; falls back to `fallback` if not a number, no upper bound when `timeoutMs <= 0`. */
export function clampYield(raw: unknown, fallback: number, timeoutMs?: number): number {
const n = typeof raw === "number" && Number.isFinite(raw) ? raw : fallback;
const lower = Math.max(n, MIN_YIELD_MS);
if (timeoutMs === undefined || timeoutMs <= 0) return lower;
return Math.min(lower, Math.max(timeoutMs - TIMEOUT_MARGIN_MS, MIN_YIELD_MS));
}
@@ -0,0 +1,174 @@
/**
* BackgroundRegistry —— generic registry for background sessions, shared by command sessions
* and subagent sessions.
*
* Responsibilities: allocating and managing background session ids, enforcing the concurrency
* cap, reclaiming idle sessions, uniform finalization when the Session/Environment ends, and a
* hard kill on process 'exit' as a fallback (JS has no destructors, so cleanup must be explicit).
* Background sessions living for days at a time are a legitimate form; idle reclamation is only
* a leak fallback — sessions unaccessed for longer than `IDLE_TTL_MS` are finalized by a
* periodic sweep.
*
* The two concurrency-cap strategies are expressed by how `makeRoom` is called:
* - Command sessions: if full at registration time, prefer evicting exited sessions, otherwise
* evict LRU (killing a background process is an acceptable cost);
* - Subagent sessions: `makeRoom` is called before launch (only evicting completed, idle ones);
* if there's no room, spawning is rejected — evicting a running subagent is equivalent to
* discarding in-progress work, which is semantically unacceptable.
*
* No lock is needed under the single-threaded event loop; but note the registry may change
* across an `await`, so check before using an entry.
* Docs: /docs/tools § "Background session caps".
*/
import { randomUUID } from "node:crypto";
/** Idle reclamation TTL (milliseconds): a session unaccessed for longer than this is treated as a leak and reclaimed. */
const IDLE_TTL_MS = 10 * 24 * 60 * 60_000; // 10 days
/** Idle reclamation check interval (milliseconds): TTL is measured in days, so an hourly sweep is sufficient. */
const IDLE_SWEEP_MS = 60 * 60_000;
/** Minimal contract a background session must satisfy to be managed by the registry. */
export interface BackgroundTask {
/** Timestamp of the most recent access (used for LRU eviction); refreshed by the registry on register/get. */
lastUsed: number;
/** Whether the session is still running (determines eviction priority). */
running: boolean;
/** Asynchronous finalization (SIGTERM -> SIGKILL / abort); idempotent. */
kill(): void;
/** Synchronous hard kill (process 'exit' fallback path: the event loop has stopped, timers are unavailable). */
killHard(): void;
}
// process 'exit' fallback: use a single module-level listener to manage all registries, avoiding
// each Session adding its own listener and triggering EventEmitter's MaxListeners warning.
const LIVE_REGISTRIES = new Set<BackgroundRegistry<BackgroundTask>>();
let exitHookInstalled = false;
function ensureExitHook(): void {
if (exitHookInstalled) return;
exitHookInstalled = true;
process.on("exit", () => {
for (const r of LIVE_REGISTRIES) r.killAllHard();
});
}
export class BackgroundRegistry<T extends BackgroundTask> {
private readonly tasks = new Map<string, T>();
private readonly idPrefix: string;
private readonly maxTasks: number;
private readonly reapTimer: ReturnType<typeof setInterval>;
private disposed = false;
constructor(opts: { idPrefix: string; maxTasks: number }) {
this.idPrefix = opts.idPrefix;
this.maxTasks = opts.maxTasks;
LIVE_REGISTRIES.add(this as unknown as BackgroundRegistry<BackgroundTask>);
ensureExitHook();
this.reapTimer = setInterval(() => this.reapIdle(), IDLE_SWEEP_MS);
this.reapTimer.unref?.();
}
get size(): number {
return this.tasks.size;
}
/**
* Makes room for a new session. Returns true immediately if not full; when full, evicts per
* `evictRunning`:
* - false (subagent): only evicts the least-recently-used **completed** session; if all are
* running, returns false (the caller rejects spawning);
* - true (command): evicts exited sessions first, otherwise LRU-evicts a running one.
*/
makeRoom(evictRunning: boolean): boolean {
if (this.tasks.size < this.maxTasks) return true;
let lruId: string | null = null;
let lruUsed = Number.POSITIVE_INFINITY;
for (const [id, t] of this.tasks) {
if (!t.running) {
this.remove(id); // Prefer evicting sessions that have already ended
return true;
}
if (t.lastUsed < lruUsed) {
lruUsed = t.lastUsed;
lruId = id;
}
}
if (!evictRunning) return false;
if (lruId) this.remove(lruId);
return this.tasks.size < this.maxTasks;
}
/**
* Registers a session, allocating and returning a unique id (`<prefix>-xxxxxxxx`). The caller
* must first free up room via `makeRoom`. `preferredSuffix` is the preferred id suffix (e.g.
* the tail of a child Session id, so the tool handle correlates with the message origin/
* frontend nesting label); falls back to random if omitted or on collision.
*/
register(task: T, preferredSuffix?: string): string {
this.ensureActive();
let id = preferredSuffix ? `${this.idPrefix}-${preferredSuffix}` : this.randomId();
while (this.tasks.has(id)) id = this.randomId();
task.lastUsed = Date.now();
this.tasks.set(id, task);
return id;
}
private randomId(): string {
return `${this.idPrefix}-${randomUUID().replace(/-/g, "").slice(0, 8)}`;
}
/** Looks up a session by id and refreshes its access time; returns undefined if not found. */
get(id: string): T | undefined {
if (this.disposed) return undefined;
const t = this.tasks.get(id);
if (t) t.lastUsed = Date.now();
return t;
}
/** Removes a session from the registry and finalizes it. */
remove(id: string): void {
const t = this.tasks.get(id);
if (!t) return;
this.tasks.delete(id);
t.kill();
}
/** Kills and clears all sessions (called when the Session/Environment ends). */
killAll(): void {
for (const t of this.tasks.values()) t.kill();
this.tasks.clear();
}
/** Synchronously hard-kills all sessions (process 'exit' fallback path). */
killAllHard(): void {
for (const t of this.tasks.values()) t.killHard();
this.tasks.clear();
}
/** Disposes: removes the fallback registration and kills all sessions. Idempotent. */
dispose(): void {
if (this.disposed) return;
this.disposed = true;
clearInterval(this.reapTimer);
LIVE_REGISTRIES.delete(this as unknown as BackgroundRegistry<BackgroundTask>);
this.killAll();
}
/** Whether the registry has been disposed (the host Session has ended). */
get isDisposed(): boolean {
return this.disposed;
}
/** Reclaims sessions idle for longer than `IDLE_TTL_MS` (leak fallback, triggered by the periodic sweep). */
private reapIdle(): void {
const cutoff = Date.now() - IDLE_TTL_MS;
for (const [id, t] of this.tasks) {
if (t.lastUsed <= cutoff) this.remove(id);
}
}
private ensureActive(): void {
if (this.disposed) {
throw new Error("background session registry disposed");
}
}
}
@@ -0,0 +1,43 @@
/**
* WakeSignal —— a single wakeup point shared by background sessions.
*
* Producer events (data arrival / run finished / new approval request) call `notify()`;
* waiters use `wait(ms)` to wait for "woken up" or expiry, whichever comes first. `notify`
* swaps in a new promise before resolving the old one, so a waiter that wakes up just
* re-checks state — it never misses an event that immediately follows.
*/
export class WakeSignal {
private promise!: Promise<void>;
private resolve!: () => void;
constructor() {
this.arm();
}
private arm(): void {
this.promise = new Promise<void>((resolve) => {
this.resolve = resolve;
});
}
/** Wakes up all waiters: swaps in a new promise before resolving the old one (avoids missing an event that immediately follows). */
notify(): void {
const r = this.resolve;
this.arm();
r();
}
/** Waits for "woken up" or `ms` to elapse, whichever comes first. */
async wait(ms: number): Promise<void> {
let timer: ReturnType<typeof setTimeout> | null = null;
const timeout = new Promise<void>((resolve) => {
// wait is part of an active operation: the timer needs to keep the process alive, and is cleaned up immediately below after notify.
timer = setTimeout(resolve, ms);
});
try {
await Promise.race([this.promise, timeout]);
} finally {
if (timer) clearTimeout(timer);
}
}
}
@@ -0,0 +1,11 @@
/**
* Barrel for the long-running command session module.
*/
export { CommandSessionManager } from "./session-manager.js";
export { ManagedSession, resultForExit } from "./session.js";
export type { ProcessExit, SpawnOptions } from "./session.js";
export {
DEFAULT_EXEC_YIELD_MS,
DEFAULT_WRITE_YIELD_MS,
DEFAULT_EMPTY_POLL_YIELD_MS,
} from "./limits.js";
@@ -0,0 +1,15 @@
/**
* Default yield durations for long-running command sessions.
*
* `yield_time_ms` is the soft budget for a tool call to "wait at most until the command ends or
* this duration elapses" (yielding on expiry is not a failure); see `../background/limits.ts`
* for the clamping logic: it only sets a floor, the ceiling is derived from the tool's own
* `timeoutMs`.
*/
/** Default wait duration (milliseconds) for `exec_command` starting a command. */
export const DEFAULT_EXEC_YIELD_MS = 60_000;
/** Default wait duration (milliseconds) for `input_command` when there's a write. */
export const DEFAULT_WRITE_YIELD_MS = 250;
/** Default wait duration (milliseconds) for `input_command` on an empty poll. */
export const DEFAULT_EMPTY_POLL_YIELD_MS = 5_000;
@@ -0,0 +1,83 @@
/**
* CommandSessionManager — registry and lifecycle management for long-running command sessions.
*
* Constructed by Environment (one per Session), injected via services and shared by the
* `exec_command` and `input_command` tools. Registry responsibilities (id allocation, concurrency
* cap, dispose, process 'exit' fallback) are handled by the generic `BackgroundRegistry` (shared
* with subagent sessions, see `../background/registry.ts`); this class only retains
* command-domain logic: spawning processes and assembling the child process environment (vault
* injection + hardening).
* Docs: /docs/tools § "Background session caps".
*/
import { ManagedSession } from "./session.js";
import { BackgroundRegistry } from "../background/index.js";
/** Concurrent managed-session cap: evicts once exceeded (exited sessions first, otherwise LRU — killing a background process has bounded cost). */
const MAX_SESSIONS = 64;
/**
* Hardening overrides applied to the child process environment: suppresses editor/credential
* prompts/pagers/color etc. that could interact, avoiding a command hanging while waiting for
* input. `GIT_EDITOR=true` prevents `git commit`/`rebase -i` from popping an editor;
* `GIT_TERMINAL_PROMPT=0` prevents git from interactively asking for credentials; in pipe mode,
* git and similar tools already auto-disable the pager, so the `PAGER` entries are just an extra
* safeguard.
*/
const HARDENED_ENV: NodeJS.ProcessEnv = {
GIT_EDITOR: "true",
GIT_TERMINAL_PROMPT: "0",
TERM: "dumb",
NO_COLOR: "1",
PAGER: "cat",
GIT_PAGER: "cat",
};
export class CommandSessionManager {
private readonly registry = new BackgroundRegistry<ManagedSession>({
idPrefix: "proc",
maxTasks: MAX_SESSIONS,
});
/** Agent vault environment variables: injected into the child process on every spawn (values never enter the model context, only the environment). */
private readonly vault: Record<string, string>;
constructor(opts?: { vault?: Record<string, string> }) {
this.vault = opts?.vault ?? {};
}
/** Starts a command, returning an **unregistered** session (no process_id yet). */
spawn(opts: { cmd: string; cwd: string }): ManagedSession {
if (this.registry.isDisposed) {
throw new Error("command session manager disposed");
}
return new ManagedSession({
cmd: opts.cmd,
cwd: opts.cwd,
// Spread order is priority: vault overrides host variables of the same name, but must
// come before HARDENED_ENV — the hardening entries (GIT_EDITOR/PAGER etc. that prevent
// interactive hangs) must never be overridable by vault.
env: { ...process.env, ...this.vault, ...HARDENED_ENV },
});
}
/** Registers a still-running session as a background process, allocating and returning a unique `process_id`. */
register(session: ManagedSession): string {
this.registry.makeRoom(true);
return this.registry.register(session);
}
/** Looks up a session by process_id and refreshes its access time; returns undefined if it doesn't exist. */
get(processId: string): ManagedSession | undefined {
return this.registry.get(processId);
}
/** Removes from the registry and cleans up the process group (called after the session exits). */
remove(processId: string): void {
this.registry.remove(processId);
}
/** Disposes: removes the fallback registration and kills all sessions (the process 'exit' fallback is hooked up by the registry itself). Idempotent. */
dispose(): void {
this.registry.dispose();
}
}
@@ -0,0 +1,228 @@
/**
* ManagedSession — runtime state and collection logic for a single command session.
*
* Spawns the process with `bash -lc <cmd>`, with stdout/stderr going through plain pipes (no
* native dependency, clean output; an interactive program that detects no TTY falls back to
* non-interactive mode, which parses more cleanly for the Agent anyway). `detached` makes the
* child process the process-group leader, so both Ctrl-C and killing the whole group rely on
* **process-group signals** (sending a signal to `-pid` also reaches background child processes).
*
* Key semantics:
* - **Termination is determined by the foreground process exiting (the exit event, waitpid
* semantics), not by waiting for stream EOF**: background child processes that inherit the
* pipe don't hold things up;
* - `collect(yieldMs)` **streams** output deltas within the budget: data is yielded as soon as it
* arrives, without waiting for the window to end; if the command exits mid-window, the trailing
* output is yielded along with it (with a capped drain window); if it's still running once the
* window expires, whatever output exists is yielded and collection ends, with the process
* switching to background; if `signal` aborts, whatever output exists is yielded and collection
* ends immediately;
* - Unread output has a cap (memory safety); when exceeded, the oldest part is dropped and
* counted, with a marker shown on read;
* - `kill()` sends SIGTERM to the process group, then SIGKILL after a grace period, reaping any
* leftover background child processes; idempotent.
*/
import { spawn, type ChildProcess } from "node:child_process";
import type { ToolResult } from "../types.js";
import { CappedTextBuffer, WakeSignal } from "../background/index.js";
/** Process-group semantics are available on POSIX; Windows falls back to signaling the child process directly. */
const SUPPORTS_PROCESS_GROUP = process.platform !== "win32";
/** Extra wait cap (ms) after the command exits to collect trailing output: enough to drain the last flush, without hanging. */
const POST_EXIT_DRAIN_MS = 50;
/** Capacity cap (characters) for a single session's unread output: prevents a chatty background process from blowing up memory. */
const OUTPUT_BUFFER_CAP = 1024 * 1024; // 1 MiB
/**
* Grace period (ms) before escalating from SIGTERM to SIGKILL: gives a process that needs to
* clean up (flush data, remove temp files) some time. The timer is unref'd, so it won't hold up
* the host process from exiting; the process-exit path sends SIGKILL directly as a fallback.
*/
const SIGKILL_GRACE_MS = 1_000;
/** Foreground process exit info. At most one of `code`/`signal` is set (consistent with Node child's exit event). */
export interface ProcessExit {
code: number | null;
signal: NodeJS.Signals | null;
}
/** Arguments required to start a command. */
export interface SpawnOptions {
/** Command string handed to `bash -lc`. */
cmd: string;
/** Working directory (absolute path). */
cwd: string;
/** Child process environment variables (the caller has already injected hardening entries like PAGER/TERM). */
env: NodeJS.ProcessEnv;
}
export class ManagedSession {
/** Timestamp of the last access (used for LRU / idle reaping). */
lastUsed: number = Date.now();
private readonly child: ChildProcess;
private readonly buffer = new CappedTextBuffer(OUTPUT_BUFFER_CAP, "earlier output");
private exited = false;
private exitInfo: ProcessExit | null = null;
private spawnError: Error | null = null;
private killed = false;
private killTimer: ReturnType<typeof setTimeout> | null = null;
// Single wake point: data arrival / process exit / spawn error all wake a waiting collect through it.
private readonly wakeSignal = new WakeSignal();
constructor(opts: SpawnOptions) {
this.child = spawn("bash", ["-lc", opts.cmd], {
cwd: opts.cwd,
env: opts.env,
detached: SUPPORTS_PROCESS_GROUP, // Become the process-group leader, so the whole group can be signaled
stdio: ["pipe", "pipe", "pipe"],
});
this.child.stdout?.setEncoding("utf8");
this.child.stderr?.setEncoding("utf8");
// stdin may already be closed by the command before input_command writes to it;
// EPIPE/ERR_STREAM_DESTROYED are an expected race and must not bubble up to the host process
// as an unhandled error.
this.child.stdin?.on("error", () => {});
this.child.stdout?.on("data", (c: string) => this.handleData(c));
this.child.stderr?.on("data", (c: string) => this.handleData(c));
// exit follows waitpid semantics: it fires as soon as bash exits, without waiting for
// stdout/stderr pipe EOF — background child processes that inherit and hold the pipe open
// won't hold up termination.
this.child.on("exit", (code, signal) => this.handleExit({ code, signal }));
this.child.on("error", (err) => this.handleError(err));
}
/** Signals the process group; ignores the case where the process/group has already exited (ESRCH). */
private signalGroup(sig: NodeJS.Signals): void {
try {
if (SUPPORTS_PROCESS_GROUP && typeof this.child.pid === "number" && this.child.pid > 0) {
process.kill(-this.child.pid, sig); // Negative pid = the whole process group
} else {
this.child.kill(sig);
}
} catch {
// ESRCH etc., ignored.
}
}
private handleData(chunk: string): void {
this.buffer.append(chunk);
this.wakeSignal.notify();
}
private handleExit(exit: ProcessExit): void {
if (this.exited) return;
this.exited = true;
this.exitInfo = exit;
this.wakeSignal.notify();
}
private handleError(err: Error): void {
if (this.exited) return;
this.spawnError = err;
this.exited = true; // A spawn failure is also treated as a terminal state
this.wakeSignal.notify();
}
/** Whether the command is still running (hasn't exited, spawn hasn't failed). */
get running(): boolean {
return !this.exited;
}
get exit(): ProcessExit | null {
return this.exitInfo;
}
get error(): Error | null {
return this.spawnError;
}
/**
* Streams output deltas within `yieldMs` (data is yielded as soon as it arrives). Once done,
* the terminal state is determined via `running`/`exit`/`error`:
* - Exits mid-window -> the trailing output is yielded along with it (extra ≤POST_EXIT_DRAIN_MS drain);
* - Still running once the window expires -> whatever output exists is yielded and collection ends, with the process switching to background;
* - `signal` aborts -> whatever output exists is yielded and collection ends immediately (the process isn't killed; the caller decides whether to keep it).
*/
async *collect(yieldMs: number, signal?: AbortSignal): AsyncGenerator<string> {
const start = Date.now();
const onAbort = (): void => this.wakeSignal.notify();
signal?.addEventListener("abort", onAbort, { once: true });
try {
// Phase one: running, data is yielded as soon as it arrives, until exit / abort / yield expires.
while (!this.exited) {
const chunk = this.buffer.drain();
if (chunk) yield chunk;
if (signal?.aborted) return;
const remaining = yieldMs - (Date.now() - start);
if (remaining <= 0) {
const tail = this.buffer.drain();
if (tail) yield tail;
return; // Still running -> yield
}
// Re-check the predicate before sleeping: data that arrives while `yield` is suspended
// wakes at a point before this wait begins, and would otherwise be missed.
if (!this.buffer.isEmpty) continue;
await this.wakeSignal.wait(remaining);
}
// Phase two: already exited (or spawn failed) -> drain the trailing output, with a cap.
const head = this.buffer.drain();
if (head) yield head;
const drainStart = Date.now();
for (;;) {
if (!this.buffer.isEmpty) {
yield this.buffer.drain();
continue;
}
const left = POST_EXIT_DRAIN_MS - (Date.now() - drainStart);
if (left <= 0) break;
await this.wakeSignal.wait(left);
if (this.buffer.isEmpty) break; // Woke with no new data (or timed out) -> draining is done
}
const tail = this.buffer.drain();
if (tail) yield tail;
} finally {
signal?.removeEventListener("abort", onAbort);
}
}
write(chars: string): void {
this.lastUsed = Date.now();
try {
if (!this.child.stdin || this.child.stdin.destroyed) return;
this.child.stdin.write(chars, () => {});
} catch {
// stdin may already be closed, ignored.
}
}
interrupt(): void {
this.lastUsed = Date.now();
this.signalGroup("SIGINT");
}
/** Closes out: sends SIGTERM to the process group, then SIGKILL after a grace period (reaping leftover background child processes); idempotent. */
kill(): void {
if (this.killed) return;
this.killed = true;
this.signalGroup("SIGTERM");
// Unconditionally escalates to SIGKILL: the foreground has exited but background child
// processes may still be around; killpg on an already-vanished group is ESRCH (harmless).
this.killTimer = setTimeout(() => this.signalGroup("SIGKILL"), SIGKILL_GRACE_MS);
this.killTimer.unref?.();
}
/** Synchronous hard kill (process 'exit' fallback: the event loop has already stopped at this point, so timers aren't available). */
killHard(): void {
this.killed = true;
if (this.killTimer) {
clearTimeout(this.killTimer);
this.killTimer = null;
}
this.signalGroup("SIGKILL");
}
}
/** Converts exit info into a tool result (the terminal marker is appended via `note`, outside the truncation, so it isn't lost with long output). */
export function resultForExit(exit: ProcessExit | null): ToolResult {
if (!exit) return { stopReason: "completed" };
if (exit.signal) return { stopReason: "failed", note: `[terminated by signal ${exit.signal}]` };
if (exit.code !== 0)
return { stopReason: "failed", note: `[exit code: ${exit.code ?? "unknown"}]` };
return { stopReason: "completed" };
}
@@ -0,0 +1,115 @@
/**
* describe_image —— image-proxy-read tool, the text-only-model variant of read_image
* (`forModel: "text-only"`, see default-config.ts): the image itself is never fed back into
* the session model (some providers flatly 400 on a tool_result carrying an image); instead it
* is sent, together with the caller-supplied `prompt`, in a single one-off request to the
* Project-configured vision model (`vision_model`), and the vision model's text answer is
* returned as the tool's output.
*
* The tool definition (description/parameters, including `prompt`) comes entirely from the
* config entry; this implementation does no runtime rewriting — which tool is used for which
* model class is declared by the entry's `forModel` annotation, so the config file is what you get.
*
* Behavioral contract (shared with read_image): `source` supports http(s) URLs and local paths;
* validation/size limits are reused from `loadImage`; on failure, outputs explanatory text and
* finishes with `failed`, never throws; on interruption, only reports `aborted`. Messages from
* the internal one-off request never enter the parent session stream (no origin, not leaked out).
* Docs: /docs/tools § "Image tools".
*/
import { imageUrlMessage, partialToolCallOutput, userText } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { LLMOutcome, ToolDefinitionConfig, VisionDescriberService } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import { formatSize, loadImage } from "./read-image.js";
/** Tool name constant (used only inside this tool module, not exposed to Environment). */
export const DESCRIBE_IMAGE_NAME = "describe_image";
/** Default question used when the caller doesn't supply a prompt. */
const DEFAULT_PROMPT =
"Describe this image in detail, including any visible text, numbers, UI elements and layout.";
/** Constructs the describe_image tool: definition (description/parameters) is taken as-is from the config entry. */
export function createDescribeImageTool(
definition: ToolDefinitionConfig,
describer: VisionDescriberService,
): BuiltinTool {
return {
name: DESCRIBE_IMAGE_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
const source = args["source"];
if (typeof source !== "string" || source.length === 0) {
yield delta('Missing required argument "source" for describe_image.');
return { stopReason: "failed" };
}
if (describer.modelId === null || describer.createLLM === undefined) {
yield delta(
"No vision model is configured for this project. The current model does not accept " +
"images; ask the user to pick a vision model in the model settings (vision_model) " +
"to enable image reading.",
);
return { stopReason: "failed" };
}
const res = await loadImage(source, ctx.workspaceDir, signal);
if (!res.ok) {
if (res.reason === "aborted") return { stopReason: "aborted" };
yield delta(res.message);
return { stopReason: "failed" };
}
const prompt =
typeof args["prompt"] === "string" && args["prompt"].trim().length > 0
? args["prompt"]
: DEFAULT_PROMPT;
const dataUrl = `data:${res.mime};base64,${res.bytes.toString("base64")}`;
// One-off vision model request: prompt + image are merged into a single user message;
// its text deltas (partial_text delta) are forwarded in real time as this tool's own
// output delta — the description streams out piece by piece, not buffered as a whole.
// Partial concatenation == the full message (see generative-model.ts), so the full text
// is not forwarded again; other messages like thinking/token_usage are ignored, never
// leaked into the parent session.
const llm = describer.createLLM();
const gen = llm.streamGenerate({
newMessages: [userText(prompt), imageUrlMessage(dataUrl)],
...(signal ? { signal } : {}),
});
yield delta(
`${res.mime}, ${formatSize(res.bytes.length)} — described by ${describer.modelId}:\n`,
);
let streamedAny = false;
let outcome: LLMOutcome | undefined;
for (;;) {
const step = await gen.next();
if (step.done) {
outcome = step.value;
break;
}
const p = step.value.payload as { type?: string; event_type?: string; text?: string };
if (p.type === "partial_text" && p.event_type === "delta" && p.text) {
streamedAny = true;
yield delta(p.text);
}
}
if (signal?.aborted) return { stopReason: "aborted" };
if (!outcome || outcome.status !== "completed") {
const detail =
outcome && "message" in outcome && outcome.message ? `: ${outcome.message}` : "";
yield delta(
`${streamedAny ? "\n" : ""}Vision model (${describer.modelId}) request ${outcome?.status ?? "failed"}${detail}`,
);
return { stopReason: "failed" };
}
if (!streamedAny) yield delta("[vision model returned no text]");
},
};
}
@@ -0,0 +1,121 @@
/**
* exec_command —— local shell executor, a built-in tool implementation (BuiltinTool).
*
* Spawns a process inside the Workspace via `bash -lc <cmd>` and streams content deltas as
* stdout/stderr chunks arrive. Waits up to `yield_time_ms`: if the command finishes in time,
* returns the full output and exit status; if it's still running when the deadline hits,
* returns the output collected so far plus a `process_id` — the process moves to background,
* managed by `CommandSessionManager`, and is interacted with afterward via `input_command`.
* Completion is decided by **the foreground
* process exiting**, not by waiting for EOF on the output stream — background children (e.g.
* `node server.js &`) that inherit the pipes won't hold up the tool.
*
* Division of responsibility with Environment (see environment.ts): this tool only produces
* content deltas; exit code/signal/spawn errors and `process_id` are reported via the return
* value's `note` (appended outside the maxOutputLength truncation, so it survives even when
* long output gets truncated). Whether it ends normally or abnormally, it always finishes via
* the return value, **never throws**; on interruption it only reports `aborted` — the
* interruption note itself is appended by Environment.
* Docs: /docs/tools § "Command sessions".
*/
import path from "node:path";
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import { DEFAULT_EXEC_YIELD_MS, resultForExit } from "./command/index.js";
import { clampYield } from "./background/index.js";
/** Tool name constant (used only inside this tool module, not exposed to Environment). */
export const EXEC_COMMAND_NAME = "exec_command";
/**
* exec_command built-in tool: parses arguments, resolves workdir, and delegates to
* `CommandSessionManager` to spawn the process and collect output.
* `definition` is overridden by Environment at construction time with the matching entry
* from ToolConfig (description/parameters/permission/limits).
* `services.commandSessions` is injected by Environment (shares the same registry with
* input_command).
*/
export function createExecCommandTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const manager = services?.commandSessions;
return {
name: EXEC_COMMAND_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
if (!manager) {
yield delta("[exec_command unavailable: no command session manager configured]");
return { stopReason: "failed" };
}
const cmd = args["cmd"];
if (typeof cmd !== "string" || cmd.length === 0) {
yield delta('Missing required argument "cmd" for exec_command.');
return { stopReason: "failed" };
}
// workdir defaults to workspaceDir; relative paths are resolved against workspaceDir.
const rawWorkdir = args["workdir"];
const workdir =
typeof rawWorkdir === "string" && rawWorkdir.length > 0
? path.resolve(ctx.workspaceDir, rawWorkdir)
: ctx.workspaceDir;
const yieldMs = clampYield(
args["yield_time_ms"],
DEFAULT_EXEC_YIELD_MS,
definition.timeoutMs,
);
// Caller already aborted: finish immediately with aborted.
if (signal?.aborted) return { stopReason: "aborted" };
let session;
try {
session = manager.spawn({ cmd, cwd: workdir });
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
yield delta(`[spawn error: ${message}]`);
return { stopReason: "failed" };
}
// On interruption, kill the whole process group (background children included) to
// avoid orphans; once the process moves to background this listener is removed in finally.
const onAbort = (): void => session.kill();
let registered = false;
signal?.addEventListener("abort", onAbort, { once: true });
try {
for await (const chunk of session.collect(yieldMs, signal)) yield delta(chunk);
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
// Still running at the deadline: register as a background process, returning
// process_id so input_command can continue accessing it.
const id = manager.register(session);
registered = true;
return {
stopReason: "completed",
note: `[process running with process_id ${id}; use input_command to send input or poll for output]`,
};
}
// Already exited: report exit status; process group cleanup (reaping any leftover
// background children) is handled uniformly in finally.
if (session.error) {
return { stopReason: "failed", note: `[spawn error: ${session.error.message}]` };
}
return resultForExit(session.exit);
} finally {
signal?.removeEventListener("abort", onAbort);
if (!registered) session.kill();
}
},
};
}
@@ -0,0 +1,109 @@
/**
* input_command — accesses a long-running command session started by `exec_command` (BuiltinTool).
*
* Finds the session by `process_id`: if `chars` is non-empty, writes it to stdin first (when it is exactly `\u0003`, special-cased as sending SIGINT to the
* process group, i.e. Ctrl-C — it must be sent alone; mixing it with other content errors out,
* since a pipe has no terminal line discipline and a mixed-in ETX byte would just be written
* into stdin silently with no effect), and if empty, nothing is written and it only polls.
* It then collects new output within `yield_time_ms` or waits for exit. If the command is still
* running, returns the same `process_id`; once exited, returns the trailing output and exit
* status and cleans up the session.
*
* Shares the same `CommandSessionManager` injected by Environment with exec_command. An
* interruption only cancels this poll — **it does not kill the background process** (the process
* was started independently earlier; interrupting one poll shouldn't kill it as a side effect).
* Docs: /docs/tools § "Command sessions".
*/
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import {
DEFAULT_EMPTY_POLL_YIELD_MS,
DEFAULT_WRITE_YIELD_MS,
resultForExit,
} from "./command/index.js";
import { clampYield } from "./background/index.js";
/** Tool name constant. */
export const INPUT_COMMAND_NAME = "input_command";
/** Ctrl-C: the ETX control character (U+0003). Received alone, it sends SIGINT to the process group instead of writing the byte into stdin. */
const INTERRUPT = String.fromCharCode(3); // U+0003 (ETX)
export function createInputCommandTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const manager = services?.commandSessions;
return {
name: INPUT_COMMAND_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
if (!manager) {
yield delta("[input_command unavailable: no command session manager configured]");
return { stopReason: "failed" };
}
const processId = args["process_id"];
if (typeof processId !== "string" || processId.length === 0) {
yield delta('Missing required argument "process_id" for input_command.');
return { stopReason: "failed" };
}
const session = manager.get(processId);
if (!session) {
yield delta(
`[input_command error: unknown process_id ${processId} (the session may have exited and been cleared)]`,
);
return { stopReason: "failed" };
}
const chars = typeof args["chars"] === "string" ? (args["chars"] as string) : "";
const empty = chars.length === 0;
const yieldMs = clampYield(
args["yield_time_ms"],
empty ? DEFAULT_EMPTY_POLL_YIELD_MS : DEFAULT_WRITE_YIELD_MS,
definition.timeoutMs,
);
if (signal?.aborted) return { stopReason: "aborted" };
// Write / interrupt (empty chars just polls). U+0003 mixed with other content errors out
// rather than being written silently (same as codex): a pipe has no terminal line
// discipline, so an ETX byte in stdin produces no interruption — the model would just
// see the command still running.
if (!empty) {
if (chars === INTERRUPT) session.interrupt();
else if (chars.includes(INTERRUPT)) {
yield delta(
'[input_command error: chars mixes U+0003 (Ctrl-C) with other content; send "\\u0003" alone to interrupt]',
);
return { stopReason: "failed" };
} else session.write(chars);
}
for await (const chunk of session.collect(yieldMs, signal)) yield delta(chunk);
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
return {
stopReason: "completed",
note: `[process still running with process_id ${processId}]`,
};
}
// Already exited: clean up the registry and report the exit status.
manager.remove(processId);
if (session.error) {
return { stopReason: "failed", note: `[spawn error: ${session.error.message}]` };
}
return resultForExit(session.exit);
},
};
}
@@ -0,0 +1,125 @@
/**
* input_subagent —— accesses a subagent session that `run_subagent` moved to the background
* (BuiltinTool).
*
* Finds the session by `subagent_id`: when `prompt` is empty, nothing is written — it just
* polls (collecting subagent messages and text deltas buffered during the background period,
* or waiting for the run to end); when non-empty and the subagent is idle, it's fed in as a new
* user message to continue on the same child Session (long-running subagent, multi-turn
* conversation); when non-empty but the subagent is still running, it errors, suggesting to
* poll first. Within the window, queued approval requests from the child session are also
* passed through (see subagent/session.ts).
*
* Difference from `input_command`: once a round of work finishes, the session is **not
* removed** (kept to receive a follow-up prompt) — it's only released when the parent Session
* ends, or evicted as an idle session once concurrency is full. Interruption only aborts this
* particular access, **it never kills the child session** (the subagent was launched
* independently earlier; the user interrupting one poll shouldn't kill it along the way).
* Docs: /docs/tools § "Subagents".
*/
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import {
DEFAULT_SUBAGENT_POLL_YIELD_MS,
DEFAULT_SUBAGENT_YIELD_MS,
resultForSubagentExit,
} from "./subagent/index.js";
import { approvalHint } from "./run-subagent.js";
import { collectWindow } from "./subagent/collect.js";
import { clampYield } from "./background/index.js";
/** Tool name constant. */
export const INPUT_SUBAGENT_NAME = "input_subagent";
export function createInputSubagentTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const manager = services?.subagentSessions;
return {
name: INPUT_SUBAGENT_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal, approve } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
if (!manager) {
yield delta("[input_subagent unavailable: no subagent session manager configured]");
return { stopReason: "failed" };
}
const subagentId = args["subagent_id"];
if (typeof subagentId !== "string" || subagentId.length === 0) {
yield delta('Missing required argument "subagent_id" for input_subagent.');
return { stopReason: "failed" };
}
const session = manager.get(subagentId);
if (!session) {
yield delta(
`[input_subagent error: unknown subagent_id ${subagentId} ` +
`(the session may have finished and been cleared)]`,
);
return { stopReason: "failed" };
}
const prompt = typeof args["prompt"] === "string" ? (args["prompt"] as string) : "";
const empty = prompt.trim().length === 0;
const yieldMs = clampYield(
args["yield_time_ms"],
empty ? DEFAULT_SUBAGENT_POLL_YIELD_MS : DEFAULT_SUBAGENT_YIELD_MS,
definition.timeoutMs,
);
if (signal?.aborted) return { stopReason: "aborted" };
// Continue with a follow-up prompt (empty prompt just polls). New input is not accepted
// while running: poll first to collect progress.
if (!empty) {
if (session.running) {
yield delta(
`[input_subagent error: subagent ${subagentId} is still running; ` +
`poll with an empty prompt to collect progress first]`,
);
return { stopReason: "failed" };
}
// startRun expresses edge cases like already-disposed via throw, collapsed here into failed (the tool never throws outward).
try {
session.startRun(prompt);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
yield delta(`[input_subagent error: ${message}]`);
return { stopReason: "failed" };
}
}
yield* collectWindow(session, {
yieldMs,
toolCallId,
...(signal ? { signal } : {}),
...(approve ? { approve } : {}),
});
// Interruption only aborts this access, it doesn't kill the child session.
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
return {
stopReason: "completed",
note: `[subagent still running with subagent_id ${subagentId}]` + approvalHint(session),
};
}
// This round of work has ended: report the end state; the session is kept (can be resumed), not removed from the registry.
const result = resultForSubagentExit(session.exit);
const idleHint = `[subagent idle with subagent_id ${subagentId}; send a follow-up prompt to continue]`;
return {
...result,
note: result.note !== undefined ? `${result.note} ${idleHint}` : idleHint,
};
},
};
}
@@ -0,0 +1,248 @@
/**
* read_image — image-reading tool, a builtin tool implementation (BuiltinTool).
*
* Reads an image and feeds it back to the model as **image content**: if `source` is an http(s)
* URL, downloads it with the global fetch (respecting the abort signal); otherwise reads it as a
* local file path (relative paths are resolved against the Workspace). Only png/jpeg/gif/webp
* are allowed (determined in order by response header / magic number / extension); errors out
* above 5MB.
*
* Division of responsibility with Environment (see environment.ts): on success, yields a brief
* descriptive delta (e.g. `image/png, 123.4 kB`), while the image itself is carried via the
* return value `ToolResult.images` (a data URL) for Environment to attach when closing out (a
* single streaming delta carries it all at once before stop, plus the final complete
* `tool_call_output`); on failure, yields explanatory text and closes with `failed`, **never
* throwing**; if interrupted, only reports `aborted` — the interruption note is appended by
* Environment.
*
* This tool is only used by sessions with a model that supports images (config entry
* `forModel: "vision"`); text-only models use describe_image instead (the image is handed to a
* configured vision model to describe, returning text — see describe-image.ts), and the image
* loading/validation logic is shared via `loadImage`.
* Docs: /docs/tools § "Image tools".
*/
import path from "node:path";
import { readFile, stat } from "node:fs/promises";
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
/** Tool name constant (used only within this tool module, never exposed to Environment). */
export const READ_IMAGE_NAME = "read_image";
/**
* Image size upper bound (bytes): errors out above this. Taken as the common denominator of
* per-provider single-image hard limits (Claude API is around 5MB, some compatible endpoints are
* lower) — since local validation passing but the next request getting a blanket 400 from the
* provider is a non-retryable path, the limit must not exceed the strictest downstream; this also
* avoids oversized images blowing up the context and Trace.
*/
export const MAX_IMAGE_BYTES = 5 * 1024 * 1024;
/** Supported image mime types (the four generally accepted across providers). */
const SUPPORTED_MIMES = new Set(["image/png", "image/jpeg", "image/gif", "image/webp"]);
/** Extension -> mime (fallback when magic-number sniffing fails). */
const EXT_TO_MIME: Record<string, string> = {
".png": "image/png",
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".gif": "image/gif",
".webp": "image/webp",
};
/** Sniffs the mime type from the file header's magic number; returns null if unrecognized. */
function sniffMime(buf: Buffer): string | null {
if (buf.length >= 8 && buf.readUInt32BE(0) === 0x89504e47) return "image/png";
if (buf.length >= 3 && buf[0] === 0xff && buf[1] === 0xd8 && buf[2] === 0xff) {
return "image/jpeg";
}
if (buf.length >= 6) {
const head = buf.subarray(0, 6).toString("latin1");
if (head === "GIF87a" || head === "GIF89a") return "image/gif";
}
if (
buf.length >= 12 &&
buf.subarray(0, 4).toString("latin1") === "RIFF" &&
buf.subarray(8, 12).toString("latin1") === "WEBP"
) {
return "image/webp";
}
return null;
}
/** Infers the mime type from a path / URL pathname's extension; returns null if it can't be inferred. */
function mimeFromExt(p: string): string | null {
return EXT_TO_MIME[path.extname(p).toLowerCase()] ?? null;
}
/** Byte count -> human-readable size (B / kB / MB, one decimal place). */
export function formatSize(bytes: number): string {
if (bytes < 1024) return `${bytes} B`;
const kb = bytes / 1024;
if (kb < 1024) return `${kb.toFixed(1)} kB`;
return `${(kb / 1024).toFixed(1)} MB`;
}
const OVERSIZE_MESSAGE = (size: number): string =>
`Image too large: ${formatSize(size)} exceeds the ${formatSize(MAX_IMAGE_BYTES)} limit.`;
const UNSUPPORTED_MESSAGE = (detected: string | null): string =>
`Unsupported image type${detected ? ` "${detected}"` : ""}: only png, jpeg, gif and webp are supported.`;
/** Result of `loadImage`: success (bytes + mime) / interrupted / failed (explanatory message). */
export type LoadImageResult =
| { ok: true; bytes: Buffer; mime: string }
| { ok: false; reason: "aborted" }
| { ok: false; reason: "failed"; message: string };
/**
* Reads and validates an image (shared by read_image and describe_image):
* an http(s) URL is downloaded with the global fetch, otherwise read as a local path (resolved
* against Workspace); validates the size upper bound and mime type (determined in order by
* response header / magic number / extension). Never throws.
*/
export async function loadImage(
source: string,
workspaceDir: string,
signal?: AbortSignal,
): Promise<LoadImageResult> {
if (signal?.aborted) return { ok: false, reason: "aborted" };
let bytes: Buffer;
let mime: string | null;
if (/^https?:\/\//i.test(source)) {
// URL branch: downloads via the global fetch (abort signal passed through to the request);
// mime is preferentially taken from the response header, falling back to magic number / URL
// extension.
let res: Response;
try {
res = await fetch(source, signal ? { signal } : {});
} catch (err) {
if (signal?.aborted) return { ok: false, reason: "aborted" };
const message = err instanceof Error ? err.message : String(err);
return {
ok: false,
reason: "failed",
message: `Failed to download image "${source}": ${message}`,
};
}
if (!res.ok) {
return {
ok: false,
reason: "failed",
message: `Failed to download image "${source}": HTTP ${res.status}`,
};
}
// When content-length is trustworthy, reject an oversized response early to avoid reading it
// into memory for nothing.
const declared = Number(res.headers.get("content-length") ?? "");
if (Number.isFinite(declared) && declared > MAX_IMAGE_BYTES) {
return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(declared) };
}
try {
bytes = Buffer.from(await res.arrayBuffer());
} catch (err) {
if (signal?.aborted) return { ok: false, reason: "aborted" };
const message = err instanceof Error ? err.message : String(err);
return {
ok: false,
reason: "failed",
message: `Failed to download image "${source}": ${message}`,
};
}
const headerMime = (res.headers.get("content-type") ?? "").split(";")[0]!.trim().toLowerCase();
let urlExtMime: string | null = null;
try {
urlExtMime = mimeFromExt(new URL(source).pathname);
} catch {
urlExtMime = null; // A URL parse failure only affects the extension fallback
}
mime = SUPPORTED_MIMES.has(headerMime) ? headerMime : (sniffMime(bytes) ?? urlExtMime);
if (mime === null && headerMime) mime = headerMime; // Include the real response type in the error
} else {
// Local-path branch: relative paths are resolved against Workspace; stat first to check the
// size before reading, to avoid reading an oversized file into memory in one go.
const filePath = path.resolve(workspaceDir, source);
try {
const st = await stat(filePath);
// Explicitly reject non-file paths such as directories: readFile's EISDIR error isn't
// model-friendly.
if (!st.isFile()) {
return {
ok: false,
reason: "failed",
message: `Failed to read image "${source}": path is not a file.`,
};
}
if (st.size > MAX_IMAGE_BYTES) {
return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(st.size) };
}
bytes = await readFile(filePath);
} catch (err) {
if (signal?.aborted) return { ok: false, reason: "aborted" };
const message = err instanceof Error ? err.message : String(err);
return {
ok: false,
reason: "failed",
message: `Failed to read image "${source}": ${message}`,
};
}
mime = sniffMime(bytes) ?? mimeFromExt(filePath);
}
if (signal?.aborted) return { ok: false, reason: "aborted" };
// Empty file/response: magic-number sniffing fails to identify it, but the extension fallback
// may still let it through — an empty base64 sent to the provider is guaranteed to error, so
// reject it here.
if (bytes.length === 0) {
return { ok: false, reason: "failed", message: `Image "${source}" is empty.` };
}
if (bytes.length > MAX_IMAGE_BYTES) {
return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(bytes.length) };
}
if (mime === null || !SUPPORTED_MIMES.has(mime)) {
return { ok: false, reason: "failed", message: UNSUPPORTED_MESSAGE(mime) };
}
return { ok: true, bytes, mime };
}
/**
* read_image builtin tool: reads a local file or downloads a URL, validates its type and size,
* then outputs a data URL image. `definition` is overridden by Environment at construction time
* with the same-named entry from ToolConfig (description/arguments/permissions/limits).
*/
export function createReadImageTool(definition: ToolDefinitionConfig): BuiltinTool {
return {
name: READ_IMAGE_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
const source = args["source"];
if (typeof source !== "string" || source.length === 0) {
yield delta('Missing required argument "source" for read_image.');
return { stopReason: "failed" };
}
const res = await loadImage(source, ctx.workspaceDir, signal);
if (!res.ok) {
if (res.reason === "aborted") return { stopReason: "aborted" };
yield delta(res.message);
return { stopReason: "failed" };
}
// Success: yield a brief one-line description as a text delta (both in the streaming and
// complete message), while the image itself is carried via the return value for
// Environment to attach.
yield delta(`${res.mime}, ${formatSize(res.bytes.length)}`);
return { images: [`data:${res.mime};base64,${res.bytes.toString("base64")}`] };
},
};
}
@@ -0,0 +1,47 @@
/**
* Built-in tool registry —— maps tool names to BuiltinTool factories.
*
* Environment uses this table to assemble entries from ToolConfig into BuiltinTool instances:
* a tool is only assembled if its name is in the table (i.e. a supported built-in tool); the
* description/parameters/permission/maxOutputLength from config are injected into the tool's
* `definition` by each factory.
* When adding a new built-in tool, just register one factory entry here — no changes to
* Environment needed.
*
* Docs: packages/docs/content/tools.{zh,en}.md (site path /docs/tools) documents every
* built-in tool and the approval flow — keep the page in sync when this table changes.
*/
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool } from "./types.js";
import { EXEC_COMMAND_NAME, createExecCommandTool } from "./exec-command.js";
import { INPUT_COMMAND_NAME, createInputCommandTool } from "./input-command.js";
import { SUBAGENT_NAME, createSubagentTool } from "./run-subagent.js";
import { INPUT_SUBAGENT_NAME, createInputSubagentTool } from "./input-subagent.js";
import { READ_IMAGE_NAME, createReadImageTool } from "./read-image.js";
import { DESCRIBE_IMAGE_NAME, createDescribeImageTool } from "./describe-image.js";
/**
* A factory that constructs a BuiltinTool instance from a tool config entry; optionally
* receives runtime services injected by Environment.
* Most tools ignore `services`; only a few (e.g. `run_subagent`) use it.
*/
export type BuiltinToolFactory = (
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
) => BuiltinTool;
/** Tool name -> factory. */
export const BUILTIN_TOOL_FACTORIES: Record<string, BuiltinToolFactory> = {
[EXEC_COMMAND_NAME]: createExecCommandTool,
[INPUT_COMMAND_NAME]: createInputCommandTool,
[SUBAGENT_NAME]: createSubagentTool,
[INPUT_SUBAGENT_NAME]: createInputSubagentTool,
[READ_IMAGE_NAME]: createReadImageTool,
// describe_image: the text-only-model variant of read_image (hands the image to the
// configured vision model for description, returns text).
// Which tool is used for which model class is declared by the config entry's forModel
// annotation; before assembly, selectBuiltinToolsForModel has already filtered out entries
// that don't apply to the session's model.
[DESCRIBE_IMAGE_NAME]: (definition, services) =>
createDescribeImageTool(definition, services?.visionDescriber ?? { modelId: null }),
};
@@ -0,0 +1,154 @@
/**
* run_subagent — delegates a subtask to a child Agent, supporting a switch to long-running
* background execution.
*
* The tool itself doesn't depend on Agent/Session, only holding an injected `SubagentRunner`
* (breaking the circular dependency). The model may freely choose the child Agent (`agent_id`)
* and model (`model_id`) via arguments; if omitted, it falls back to reusing the current Agent
* and the Project's default Model respectively. The spawned child session is managed by
* `ManagedSubagentSession` (sharing the `SubagentSessionManager` injected by Environment with
* `input_subagent`).
*
* The two-phase semantics mirror `exec_command`: within the `yield_time_ms` window, child-session
* messages are forwarded live (tagged with origin, so the frontend can see the child Agent's tool
* calls and token usage) and the child Agent's text deltas are copied as this tool's output; if
* the child Agent finishes within the window, its terminal state is returned and the child
* session is released; if it's still running once the window expires, it's registered as a
* background session, returning `subagent_id` for subsequent access the same way as
* `input_command` (polling / appending a Prompt, see input-subagent.ts).
*
* Approval: `run_subagent` itself is a read-write tool (`rw`), so its invocation requires Human
* approval; the child session's tool approval requests are forwarded to the same Human within
* the window via the session's approval queue (tagged with origin), and queued for the next
* access while running in the background. An interruption within the startup window kills the
* child session per exec_command semantics; precheck errors such as exceeding the depth limit or
* a nonexistent agent are expressed by the runner as a throw, and collapsed to failed.
* Docs: /docs/tools § "Subagents".
*/
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import {
DEFAULT_SUBAGENT_YIELD_MS,
ManagedSubagentSession,
resultForSubagentExit,
} from "./subagent/index.js";
import { collectWindow } from "./subagent/collect.js";
import { clampYield } from "./background/index.js";
/** Tool name constant (used only within this tool module, never exposed to Environment). */
export const SUBAGENT_NAME = "run_subagent";
/** Pending-approval hint: lets the model know it should poll again to move the child Agent forward. */
export function approvalHint(session: ManagedSubagentSession): string {
const n = session.pendingApprovals;
return n > 0 ? ` [subagent is waiting for approval of ${n} tool call(s); poll to review]` : "";
}
/** Builds run_subagent's BuiltinTool from tool config + injected services. */
export function createSubagentTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const runner = services?.subagentRunner;
const manager = services?.subagentSessions;
return {
name: SUBAGENT_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal, approve } = ctx;
const fail = function* (msg: string): Generator<OmniMessage> {
yield partialToolCallOutput({ eventType: "delta", output: msg, toolCallId });
};
// Missing arguments / unconfigured services both collapse to an explanatory output rather
// than throwing (consistent with other tools).
if (!runner) {
yield* fail("[run_subagent unavailable: no subagent runner configured]");
return { stopReason: "failed" };
}
if (!manager || manager.isDisposed) {
yield* fail("[run_subagent unavailable: no subagent session manager available]");
return { stopReason: "failed" };
}
const prompt = typeof args.prompt === "string" ? args.prompt : "";
if (prompt.trim().length === 0) {
yield* fail("[run_subagent error: missing required string argument `prompt`]");
return { stopReason: "failed" };
}
const agentId = typeof args.agent_id === "string" ? args.agent_id : undefined;
const modelId = typeof args.model_id === "string" ? args.model_id : undefined;
const yieldMs = clampYield(
args.yield_time_ms,
DEFAULT_SUBAGENT_YIELD_MS,
definition.timeoutMs,
);
if (signal?.aborted) return { stopReason: "aborted" };
// Concurrency cap (running child Agents are never evicted): reject spawning if there's no
// room.
if (!manager.makeRoom()) {
yield* fail(
"[run_subagent error: too many background subagents; poll or finish existing ones first]",
);
return { stopReason: "failed" };
}
// Spawn the child Session (precheck errors such as exceeding the depth limit or a
// nonexistent agent are expressed as a throw).
let session: ManagedSubagentSession;
try {
const handle = await runner.spawn({
...(agentId !== undefined ? { agentId } : {}),
...(modelId !== undefined ? { modelId } : {}),
});
session = new ManagedSubagentSession(handle);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
yield* fail(`[run_subagent error: ${message}]`);
return { stopReason: "failed" };
}
// An interruption within the startup window kills the child session (consistent with
// exec_command); once switched to background, this listener is removed in `finally`.
const onAbort = (): void => session.kill();
let registered = false;
signal?.addEventListener("abort", onAbort, { once: true });
try {
session.startRun(prompt);
yield* collectWindow(session, {
yieldMs,
toolCallId,
...(signal ? { signal } : {}),
...(approve ? { approve } : {}),
});
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
// Still running once the window expires: register as a background session, returning
// subagent_id for input_subagent to continue accessing it.
const id = manager.register(session);
registered = true;
return {
stopReason: "completed",
note:
`[subagent running with subagent_id ${id}; use input_subagent to poll for progress ` +
`or send a follow-up prompt]` +
approvalHint(session),
};
}
// Finished within the window: report the terminal state; releasing the child session is
// handled uniformly in `finally` (never registered, so no subagent_id).
return resultForSubagentExit(session.exit);
} finally {
signal?.removeEventListener("abort", onAbort);
if (!registered) session.kill();
}
},
};
}
@@ -0,0 +1,52 @@
/**
* collectWindow —— yield-window collector shared by run_subagent / input_subagent.
*
* Within the `yieldMs` window, emits child-session output in real time: buffered child-session
* messages (already origin-tagged, passed through to the frontend) and subagent text deltas
* (fed back to the LLM as this parent tool's own output delta). The window also hooks up an
* approval outlet, forwarding the child session's queued approval requests one by one to the
* Human via `approve`. The window ends on "run finished / signal abort / deadline reached", and
* does a final drain right before ending (to catch the tail buffer at the moment the run
* finishes). Deciding the end state and finalizing are the caller's responsibility.
*/
import { partialToolCallOutput } from "../../../omnimessage/index.js";
import type { OmniMessage } from "../../../omnimessage/index.js";
import type { ApproveFn } from "../../../interfaces.js";
import type { ManagedSubagentSession } from "./session.js";
export async function* collectWindow(
session: ManagedSubagentSession,
opts: { yieldMs: number; toolCallId: string; signal?: AbortSignal; approve?: ApproveFn },
): AsyncGenerator<OmniMessage> {
const { yieldMs, toolCallId, signal, approve } = opts;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
const detach = approve ? session.attachApprovalSink(approve) : null;
// abort only ends this window (whether to kill the child session is up to the caller);
// wakes up a pending waitWake so it returns immediately.
const onAbort = (): void => session.wakeup();
signal?.addEventListener("abort", onAbort, { once: true });
try {
const start = Date.now();
for (;;) {
for (const m of session.drainMessages()) yield m;
const text = session.drainText();
if (text) yield delta(text);
if (!session.running) break;
if (signal?.aborted) break;
const remaining = yieldMs - (Date.now() - start);
if (remaining <= 0) break;
// Re-check the predicate before sleeping: output arriving while `yield` is suspended
// would fire its wakeup before this wait even starts, and get missed otherwise.
if (session.hasPending) continue;
await session.waitWake(remaining);
}
// Final drain: there may still be a tail buffer right when the run finishes/yields.
for (const m of session.drainMessages()) yield m;
const tail = session.drainText();
if (tail) yield delta(tail);
} finally {
signal?.removeEventListener("abort", onAbort);
detach?.();
}
}
@@ -0,0 +1,7 @@
/**
* Barrel for the background subagent session module.
*/
export { SubagentSessionManager } from "./session-manager.js";
export { ManagedSubagentSession, resultForSubagentExit } from "./session.js";
export type { SubagentExit } from "./session.js";
export { DEFAULT_SUBAGENT_YIELD_MS, DEFAULT_SUBAGENT_POLL_YIELD_MS } from "./limits.js";
@@ -0,0 +1,11 @@
/**
* Default yield duration for background subagent sessions.
*
* See `../background/limits.ts` for the clamping logic: only a lower bound is set, and the
* upper bound is derived from the tool's own `timeoutMs`.
*/
/** Default wait duration (ms) for `run_subagent` launching a task and `input_subagent` appending a Prompt to continue. */
export const DEFAULT_SUBAGENT_YIELD_MS = 300_000;
/** Default wait duration (ms) for `input_subagent` empty polling. */
export const DEFAULT_SUBAGENT_POLL_YIELD_MS = 10_000;
@@ -0,0 +1,63 @@
/**
* SubagentSessionManager —— registry and lifecycle management for background subagent sessions.
*
* Constructed by Environment (one per Session), injected via services to be shared by the
* `run_subagent` and `input_subagent` tools. Registry duties are handled by the generic
* `BackgroundRegistry` (shared with command sessions, see `../background/registry.ts`).
* Difference from command sessions: when at capacity, **running sessions are never evicted**
* (discarding in-progress subagent work is unacceptable) — only completed, idle ones are
* evicted; if there's still no room, the tool rejects spawning a new one.
* Docs: /docs/tools § "Background session caps".
*/
import { BackgroundRegistry } from "../background/index.js";
import type { ManagedSubagentSession } from "./session.js";
/**
* Cap on concurrently managed background subagent sessions. This is a **spawn admission cap**,
* not a hard limit: there's an await between the `makeRoom` check (before spawn) and `register`
* (after the yield window ends), so parallel run_subagent calls can briefly push the registered
* count over the cap — an already-running child session is never discarded just to hold the line.
*/
const MAX_SESSIONS = 8;
export class SubagentSessionManager {
private readonly registry = new BackgroundRegistry<ManagedSubagentSession>({
idPrefix: "subagent",
maxTasks: MAX_SESSIONS,
});
/** Whether the manager has been disposed (the host Session has ended). */
get isDisposed(): boolean {
return this.registry.isDisposed;
}
/** Whether there's still room for a new background session (evicting a completed, idle one if needed; never evicts a running one). */
makeRoom(): boolean {
return this.registry.makeRoom(false);
}
/**
* Registers a still-running session as a background session, allocating and returning a
* unique `subagent_id`: `subagent-<last 8 hex of child Session id>` (falls back to random on
* collision), whose suffix aligns with the message origin/frontend nesting label
* (`agent-<last 3 chars>`) for correlation.
*/
register(session: ManagedSubagentSession): string {
// A full yield window has elapsed since the pre-spawn makeRoom check, so the registry may
// have been filled by parallel calls in the meantime: free up room once more (only evicting
// completed, idle ones); if still no room, register anyway, tolerating a brief overshoot
// (see MAX_SESSIONS).
this.registry.makeRoom(false);
return this.registry.register(session, session.sessionId.slice(-8));
}
/** Looks up a session by subagent_id and refreshes its access time; returns undefined if not found. */
get(subagentId: string): ManagedSubagentSession | undefined {
return this.registry.get(subagentId);
}
/** Disposes: removes the fallback registration and finalizes all sessions (the process 'exit' fallback is hooked by the registry itself). Idempotent. */
dispose(): void {
this.registry.dispose();
}
}
@@ -0,0 +1,304 @@
/**
* ManagedSubagentSession — a subagent session capable of running in the background.
*
* Holds a `SubagentHandle` and drives its `run` (pump): the same child Session may run across
* multiple rounds (the first round is initiated by `run_subagent`, later rounds append a Prompt
* via `input_subagent`). Structurally mirrors a command session (ManagedSession): the parent tool
* call collects output live within the yield window; output produced outside the window (while
* running in the background) goes into a buffer, delivered all at once on the next access.
*
* Three kinds of output, each with its own destination:
* - **Message buffer**: all of the child session's OmniMessage (already tagged with origin), for
* the parent tool call to forward to the frontend for rendering; capped in count, overflow
* drops the oldest (only affects frontend replay — the child Session's own Trace loses no
* data);
* - **Text buffer**: assistant text deltas from the direct child layer (origin one hop), fed back
* as the parent tool's own output to the LLM; capped in capacity (prevents memory bloat),
* overflow drops the oldest with a marker;
* - **Approval queue**: the child session's tool approval requests. While running in the
* background, the parent session may have no active tool call to forward approval through, so
* the request is queued and the child session blocks waiting; the parent tool call
* (run_subagent / input_subagent) attaches an approval sink (`attachApprovalSink`) within its
* window to consult Human one request at a time — if detached mid-consultation (window ends),
* the request stays queued, and a late-arriving decision still takes effect (settled guard,
* first to arrive wins).
*
* Cleanup: `kill()` aborts the current run via AbortSignal, denies all pending approvals, and
* releases child Session resources; idempotent. The child Session runs in-process, so there's no
* need for a separate synchronous hard-kill path (its command sessions are reaped by their own
* exit fallback); `killHard` is equivalent to `kill`.
*/
import type { OmniMessage } from "../../../omnimessage/index.js";
import type { ApprovalDecision, ToolCallPayload } from "../../../omnimessage/index.js";
import type { ApproveFn, SubagentHandle } from "../../../interfaces.js";
import type { ToolResult } from "../types.js";
import { CappedTextBuffer, WakeSignal } from "../background/index.js";
/** Message buffer count cap: overflow drops the oldest (only frontend replay is affected — the child Session has its own Trace). */
const MESSAGE_BUFFER_CAP = 4096;
/** Text buffer capacity cap (characters): prevents a chatty child Agent from blowing up memory. */
const OUTPUT_BUFFER_CAP = 1024 * 1024; // 1 MiB
/** Terminal state of one run. */
export interface SubagentExit {
status: "completed" | "failed";
note?: string;
}
/** A pending approval request: the settled guard makes the decision first-to-arrive-wins (late/duplicate decisions are ignored). */
interface PendingApproval {
toolCall: OmniMessage<ToolCallPayload>;
settled: boolean;
resolve: (decision: ApprovalDecision) => void;
}
export class ManagedSubagentSession {
/** Timestamp of the last access (used for the eviction policy). */
lastUsed: number = Date.now();
private readonly handle: SubagentHandle;
private readonly abortCtrl = new AbortController();
private messages: OmniMessage[] = [];
private readonly textBuffer = new CappedTextBuffer(OUTPUT_BUFFER_CAP, "earlier subagent output");
private isRunning = false;
private exitInfo: SubagentExit | null = null;
private killed = false;
private readonly approvals: PendingApproval[] = [];
private sink: { approve: ApproveFn; detached: Promise<void> } | null = null;
private sinkEpoch = 0;
private pumpingApprovals = false;
// Single wake point: new message / run finished / new approval request all wake a waiting waitWake through it.
private readonly wakeSignal = new WakeSignal();
constructor(handle: SubagentHandle) {
this.handle = handle;
}
/** Child Session id (one hop of a message's origin); `subagent_id` is derived from its tail so the frontend can correlate it. */
get sessionId(): string {
return this.handle.sessionId;
}
/** Whether a round of the task is currently running. */
get running(): boolean {
return this.isRunning;
}
/** Terminal state of the most recent run; null if no round has ever completed. */
get exit(): SubagentExit | null {
return this.exitInfo;
}
/** Number of pending approval requests (the parent tool uses this to hint the model to poll again). */
get pendingApprovals(): number {
return this.approvals.length;
}
/** Whether there's unread output (buffered messages or text); used to re-check the predicate before waiting (see collect.ts). */
get hasPending(): boolean {
return this.messages.length > 0 || !this.textBuffer.isEmpty;
}
/**
* Starts a new round of the task on the child Session (async pump, doesn't block the caller).
* Throws if already disposed or still running (converted to an explanatory output by the
* caller).
*/
startRun(prompt: string): void {
if (this.killed) throw new Error("subagent session disposed");
if (this.isRunning) throw new Error("subagent is still running");
this.isRunning = true;
this.exitInfo = null;
void this.pump(prompt);
}
/** Takes the buffered child-session messages (already tagged with origin, for the parent tool to forward). */
drainMessages(): OmniMessage[] {
if (this.messages.length === 0) return [];
const out = this.messages;
this.messages = [];
return out;
}
/** Takes the currently unread child Agent text (including the drop marker); clears the buffer. */
drainText(): string {
return this.textBuffer.drain();
}
/** External wakeup (e.g. the parent tool call was aborted): makes a waiting `waitWake` return immediately. */
wakeup(): void {
this.wakeSignal.notify();
}
/** Waits for "woken up" or `ms` to expire, whichever comes first. */
async waitWake(ms: number): Promise<void> {
await this.wakeSignal.wait(ms);
}
/**
* Attaches an approval sink: the parent tool call active within the window hands in its own
* `ctx.approve`, and queued approval requests are consulted with Human through it one at a
* time. Returns a detach function (called when the window ends); a later attach replaces the
* former one.
*/
attachApprovalSink(approve: ApproveFn): () => void {
const epoch = ++this.sinkEpoch;
let onDetach!: () => void;
const detached = new Promise<void>((resolve) => {
onDetach = resolve;
});
this.sink = { approve, detached };
void this.pumpApprovals();
return () => {
if (this.sinkEpoch === epoch) this.sink = null;
onDetach();
};
}
/** Cleanup: aborts the current run, denies pending approvals, releases child Session resources; idempotent. */
kill(): void {
if (this.killed) return;
this.killed = true;
this.abortCtrl.abort();
for (const req of [...this.approvals]) this.settle(req, "deny");
// If running, released by pump's finally after it finishes; otherwise released immediately.
if (!this.isRunning) this.handle.dispose();
this.wakeSignal.notify();
}
/** Synchronous hard-kill path: the child Session runs in-process with no separate OS resources, so this is equivalent to `kill`. */
killHard(): void {
this.kill();
}
// -------------------------------------------------------------------------
// Internal: pump and buffering
// -------------------------------------------------------------------------
/** Drives one round of `handle.run`: buffers messages and text, settling the terminal state when it ends. */
private async pump(prompt: string): Promise<void> {
let wroteAny = false;
let childAbort: string | null = null;
try {
for await (const msg of this.handle.run({
prompt,
signal: this.abortCtrl.signal,
approve: this.childApprove,
})) {
this.bufferMessage(msg);
if ((msg.origin?.length ?? 0) === 1) {
const p = msg.payload as {
type?: string;
event_type?: string;
text?: string;
reason?: string;
};
// A direct child layer's abort event: the child session was interrupted/failed (LLM
// request error, user interruption, etc). A child session failure doesn't throw, it
// only emits an event, based on which this round is reported as failed rather than
// marked completed.
if (p.type === "abort") {
childAbort = p.reason ?? "aborted";
} else if (
p.type === "partial_text" &&
p.event_type === "delta" &&
typeof p.text === "string" &&
p.text
) {
wroteAny = true;
this.appendText(p.text);
}
}
this.wakeSignal.notify();
}
if (childAbort !== null) {
this.exitInfo = { status: "failed", note: `[subagent aborted: ${childAbort}]` };
} else if (!wroteAny) {
this.exitInfo = {
status: "completed",
note: "[subagent finished without a text answer]",
};
} else {
this.exitInfo = { status: "completed" };
}
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
this.exitInfo = { status: "failed", note: `[subagent error: ${message}]` };
} finally {
this.isRunning = false;
if (this.killed) this.handle.dispose();
this.wakeSignal.notify();
}
}
private bufferMessage(msg: OmniMessage): void {
this.messages.push(msg);
// Overflow drops the oldest: only affects frontend replay — the child Session's Trace and text buffer are unaffected.
if (this.messages.length > MESSAGE_BUFFER_CAP) this.messages.shift();
}
private appendText(text: string): void {
this.textBuffer.append(text);
}
// -------------------------------------------------------------------------
// Internal: approval queue
// -------------------------------------------------------------------------
/** Approval callback handed to the child Session: the request is queued and waits for some parent tool call to consult Human and give a decision. */
private readonly childApprove: ApproveFn = (toolCall) => {
if (this.killed) return Promise.resolve("deny");
return new Promise<ApprovalDecision>((resolve) => {
this.approvals.push({ toolCall, settled: false, resolve });
this.wakeSignal.notify(); // Wake the parent tool call waiting within the window, so it can consult as soon as possible
void this.pumpApprovals();
});
};
/** Settles an approval decision: first to arrive wins, late/duplicate decisions are ignored. */
private settle(req: PendingApproval, decision: ApprovalDecision): void {
if (req.settled) return;
req.settled = true;
const idx = this.approvals.indexOf(req);
if (idx >= 0) this.approvals.splice(idx, 1);
req.resolve(decision);
this.wakeSignal.notify();
}
/**
* Hands the request at the head of the queue to the currently attached approval sink, one at a
* time. Stops when the window ends (the sink is detached); unresolved requests stay queued for
* the next sink; if a consultation already in flight resolves late, the decision still takes
* effect via settle.
*/
private async pumpApprovals(): Promise<void> {
if (this.pumpingApprovals) return;
this.pumpingApprovals = true;
try {
while (this.sink && this.approvals.length > 0) {
const sink = this.sink;
const req = this.approvals[0]!;
const answer = sink.approve(req.toolCall).then(
(d) => this.settle(req, d),
() => this.settle(req, "deny"), // An approval sink error is treated as a denial (avoids leaving the child session stuck forever)
);
await Promise.race([answer, sink.detached]);
if (req.settled) continue;
if (this.sink && this.sink !== sink) continue; // The sink was replaced by a new call: retry with the new sink
break; // The sink was detached and still unresolved: stay queued for the next sink
}
} finally {
this.pumpingApprovals = false;
}
}
}
/** Converts a run's terminal state into a tool result (note is appended outside the truncation, so it isn't lost with long output). */
export function resultForSubagentExit(exit: SubagentExit | null): ToolResult {
if (!exit) return { stopReason: "completed" };
return { stopReason: exit.status, ...(exit.note !== undefined ? { note: exit.note } : {}) };
}
@@ -0,0 +1,79 @@
/**
* BuiltinTool abstraction — lets Environment avoid special-casing any specific tool name.
*
* Each builtin tool carries its own: `name`, the `definition` handed to the LLM, and a streaming
* `execute`. Environment dispatches purely by looking up `name`; an unknown tool collapses to an
* explanatory `tool_call_output`, never throwing. Adding a new tool later (e.g. file read/write,
* retrieval) only requires implementing this interface and registering it with the registry, with
* no changes needed to Environment.
*/
import type { OmniMessage, StopReason } from "../../omnimessage/index.js";
import type { ApproveFn, ToolDefinitionConfig } from "../../interfaces.js";
/**
* Tool execution context: runtime information needed to execute one tool call.
* Docs: /docs/tools § "Execution contract".
*/
export interface ToolExecutionContext {
/** Workspace absolute path; relative-path arguments should be resolved against it. */
workspaceDir: string;
/** The tool_call_id passed through unchanged, used to build streaming deltas and nested origin tags. */
toolCallId: string;
/** Abort signal; the tool should close out and return as soon as possible once it fires. */
signal?: AbortSignal;
/** The parent Agent's approval callback; run_subagent passes it through to the child Session so it inherits the parent's approval mode (unused by most tools). */
approve?: ApproveFn;
}
/**
* Tool execution result (the generator's return value); treated as `completed` if omitted.
* Docs: /docs/tools § "Execution contract".
*/
export interface ToolResult {
stopReason?: StopReason;
/**
* Terminal marker (e.g. `[exit code: 1]`): appended by Environment during its unified
* close-out, **outside** the maxOutputLength truncation, and streamed to the frontend as an
* extra chunk — so the failure marker isn't lost when long output gets truncated (it would be
* cut off if produced as a content delta instead).
*/
note?: string;
/**
* Images carried by the tool output (e.g. an image read by read_image): each entry is a
* `data:<mime>;base64,...` data URL. Attached by Environment during close-out: a single
* streaming delta carries it all at once before stop, plus the final complete
* `tool_call_output` (only carried on normal completion; images are not chunked and don't
* count toward text truncation).
*/
images?: string[];
}
/**
* Builtin tool interface. `execute` receives the already-parsed tool argument object and the
* execution context, streaming out OmniMessage as an async generator. Contract (a relaxed
* version — framing and close-out are handled uniformly by Environment):
*
* - **Own output**: yielding the **delta** of `partial_tool_call_output` is enough; `start`/`stop`
* are optional (Environment ignores the tool's start/stop and frames it itself), and there's
* **no need** to produce a complete `tool_call_output` either (the complete message,
* maxOutputLength forward truncation, and close-out are all derived by Environment from the
* deltas). If a tool does produce a complete `tool_call_output` anyway, Environment uses it as
* the basis for content and stop reason (tolerated for compatibility, not recommended).
* - **Nested forwarding**: yielding any message **tagged with origin** is passed through by
* Environment unchanged (e.g. run_subagent forwarding all of a child session's messages).
* - **Stop reason**: reported via the generator's return value (defaults to completed); a throw
* is collapsed by Environment into aborted/failed based on interruption/error, never
* propagating up as an exception.
* Docs: /docs/interfaces § "The inner tool contract: BuiltinTool"; /docs/tools § "Execution contract".
*/
export interface BuiltinTool {
/** Tool name (corresponds to the tool_call.name returned by the LLM). */
name: string;
/** Tool definition handed to the LLM (including description / parameters / permission / maxOutputLength). */
definition: ToolDefinitionConfig;
/** Executes one tool call: args is the already-parsed argument object, ctx is the runtime context. */
execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void>;
}
+45
View File
@@ -0,0 +1,45 @@
/**
* @prismshadow/penguin-core — public entry point for the PenguinHarness core SDK.
*
* Exports the OmniMessage protocol, the three interface contracts (Human/LLM/Environment),
* and the runtime entry points for Agent / Session / context_engine along with their
* submodules (state / llm / environment / trace).
*
* Typical usage:
*
* ```ts
* const agent = await createAgent({ agentId: "default_agent" });
* const session = await agent.createSession({ workspaceDir, modelId });
* for await (const output of session.run([userText("...")])) { ... }
* ```
*/
// Protocol and interface contracts (foundation)
export * from "./omnimessage/index.js";
export * from "./interfaces.js";
// Submodules
export * from "./state/index.js";
export * from "./llm/index.js";
export * from "./environment/index.js";
export * from "./trace/index.js";
// Runtime entry points
export { ContextEngine } from "./engine/context-engine.js";
export type {
CompactAvailability,
CompactionSettings,
ContextEngineDeps,
EngineInitialState,
RunOptions,
TraceSink,
} from "./engine/context-engine.js";
export { Session } from "./session.js";
export type { SessionConfig } from "./session.js";
export { buildTitlePrompt, generateTitleWithLLM, sanitizeTitle } from "./session-title.js";
export type { SessionTitleResult } from "./session-title.js";
export { Agent, createAgent } from "./agent.js";
export type { CreateAgentOptions, CreateSessionOptions, ResumeSessionOptions } from "./agent.js";
/** SDK version number. */
export const VERSION = "0.0.1";
+279
View File
@@ -0,0 +1,279 @@
/**
* Internal SDK interface contracts: LLM, Environment.
*
* `context_engine` only handles OmniMessage; protocol conversion and concrete implementations
* are each interface's own responsibility.
* Human is not an "interface/class with methods" but the SDK's input/output boundary itself:
* output is streamed by `Session.run()` as an async generator, and input is delivered via
* `run`'s `RunOptions` — approvals are requested one at a time through the injected `approve`
* callback, and interruption goes through `signal`. Hence no Human interface is defined here.
*
* These types form the foundational contract shared by all units; implementing units integrate
* against them.
*
* Docs: packages/docs/content/interfaces.{zh,en}.md (site path /docs/interfaces) explains each
* contract and its extension seams — keep the page in sync when changing signatures here.
*/
import type {
ApprovalDecision,
OmniMessage,
StopReason,
ToolCallPayload,
ToolDefinition,
} from "./omnimessage/types.js";
// Concrete classes, used only for EnvironmentServices type annotations (type-only import; no runtime dependency, no circular reference).
import type { CommandSessionManager } from "./environment/tools/command/session-manager.js";
import type { SubagentSessionManager } from "./environment/tools/subagent/session-manager.js";
import type { ToolCallIdAllocator } from "./llm/tool-call-ids.js";
// ---------------------------------------------------------------------------
// Tool definitions and configuration
// ---------------------------------------------------------------------------
// ToolDefinition is defined in omnimessage/types.ts (session_meta embeds the full tool schema directly); re-exported here to keep the original import path.
export type { ToolDefinition } from "./omnimessage/types.js";
/** Tool permission: read-only / read-write. */
export type ToolPermission = "r" | "rw";
/**
* Runtime configuration for a single tool.
* Docs: /docs/tools § "Configuration fields".
*/
export interface ToolDefinitionConfig {
name: string;
description: string;
parameters?: Record<string, unknown>;
permission?: ToolPermission;
/**
* Which class of session model this entry targets: `"vision"` only for models that support
* images (e.g. read_image), `"text-only"` only for text-only models (e.g. describe_image);
* omitted means available for all models. Filtered by session model at assembly time
* (see `selectBuiltinToolsForModel`).
*/
forModel?: "vision" | "text-only";
/** Timeout for a single tool call (ms); on timeout, ends as `failed`; <=0 disables it. */
timeoutMs?: number;
/** Max length of tool output; Environment truncates from the front (keeping the head) if exceeded; <=0 disables it. */
maxOutputLength?: number;
}
export interface MCPServerConfig {
name: string;
config: Record<string, unknown>;
}
/** Set of tool configs required to initialize Environment. */
export interface ToolConfig {
customTools: ToolDefinitionConfig[];
mcpServers: MCPServerConfig[];
}
/**
* Per-tool approval callback: the Human boundary gives allow/deny for each complete `tool_call`.
* `context_engine` calls it once per tool call within a turn. Subagents forward the parent's
* approval callback, so the child Agent **inherits the parent Agent's approval mode**.
* Docs: /docs/interfaces § "ApproveFn".
*/
export type ApproveFn = (toolCall: OmniMessage<ToolCallPayload>) => Promise<ApprovalDecision>;
// ---------------------------------------------------------------------------
// LLM interface
// ---------------------------------------------------------------------------
export type ThinkingLevelName = "none" | "low" | "medium" | "high" | "xhigh";
/**
* GenerativeModel initialization config.
* Docs: /docs/interfaces § "GenerativeModelConfig".
*/
export interface GenerativeModelConfig {
modelId: string;
apiKey?: string;
baseUrl?: string;
/**
* AgentHub client protocol (`openai` / `claude-4-8` / `deepseek-v4` / …). If omitted, AgentHub
* infers it from `modelId`; custom-named models or third-party models using the OpenAI protocol
* must specify it explicitly.
*/
clientType?: string;
tools: ToolDefinition[];
/** Full system Prompt after placeholder substitution in the system_config.system_prompt template. */
systemPrompt?: string;
contextWindow?: number;
maxTokens?: number;
thinkingLevel?: ThinkingLevelName;
/** LLM Request timeout (ms): from system_config.model.timeoutMs; <=0 disables it. Defaults to 120000. */
requestTimeoutMs?: number;
/**
* tool_call_id uniqueness registry (Session-level). Pass the same instance when rebuilding a new
* GenerativeModel on compaction so the uniqueness scope covers the whole Session; defaults to a fresh
* one. See llm/tool-call-ids.ts.
*/
toolCallIds?: ToolCallIdAllocator;
}
export interface GenerativeModelParameters {
/** OmniMessage array for the input newly added this turn; implementations must merge it into a single UniMessage (multiple roles not accepted). */
newMessages: OmniMessage[];
signal?: AbortSignal;
}
/**
* The terminal state of an LLM request, returned as the **return value** of the `streamGenerate`
* async generator (not a yielded message). The status values share the same five-value protocol
* as OmniMessage `stop_reason`:
* - `completed`: finished normally (already produced `token_usage`);
* - `timeout`: LLM timed out or lost connection, needs reconnect — retried by `context_engine`
* within the same run;
* - `malformed`: AgentHub response failed JSON parsing, needs reconnect — also retried by
* `context_engine`;
* - `aborted`: user-initiated interruption — stop and hand back to the user;
* - `failed`: other non-retryable errors (auth/params, etc.) — stop and hand back to the user
* (`message` provides the display text).
* Docs: /docs/interfaces § "LLMOutcome semantics".
*/
export interface LLMOutcome {
status: StopReason;
message?: string;
}
/**
* A stateful LLM object attached to a Session.
* `streamGenerate` yields streaming `partial_*` messages as an async generator, and appends the
* corresponding complete `model_msg` once each fragment ends; Token usage is emitted as a
* `token_usage` event_msg. **Never throws to `context_engine`**: any interruption/exception is
* closed off in well-formed structure and returned normally, and **must** report the terminal
* state via `LLMOutcome` — error handling happens entirely inside the LLM interface, and
* `context_engine` only decides subsequent actions based on the outcome.
* Docs: /docs/interfaces § "LLMInterface".
*/
export interface LLMInterface {
streamGenerate(parameters: GenerativeModelParameters): AsyncGenerator<OmniMessage, LLMOutcome>;
}
// ---------------------------------------------------------------------------
// Environment interface
// ---------------------------------------------------------------------------
/**
* Handle for a child Agent session: derived by `SubagentRunner.spawn`,
* representing a child Session that can run over multiple turns. Deriving (spawn) is separate
* from running (run), so the same child Session can accept an additional Prompt and keep running
* after a turn ends (a long-running subagent, accessed via `input_subagent`).
* Docs: /docs/interfaces § "Subagent interfaces".
*/
export interface SubagentHandle {
/** The child Session's id: the origin hop of messages produced by run; `subagent_id` is derived from its tail for the frontend to correlate. */
sessionId: string;
/**
* Runs one turn of a task on the child Session. Emitted child-session messages **all already
* carry the origin marker** (the child Session id); the first message of the first run is the
* child Session's `session_meta`, and tool_calls received by the forwarded approval callback
* carry origin as well.
*/
run(input: {
/** The task Prompt handed to the child Agent. */
prompt: string;
signal?: AbortSignal;
/** The parent Agent's approval callback; forwarded to the child Session to inherit the parent's approval mode. */
approve?: ApproveFn;
}): AsyncGenerator<OmniMessage>;
/** Releases runtime resources held by the child Session (e.g. its managed command sessions). Idempotent. */
dispose(): void;
}
/**
* Child Agent runner: injected into the `run_subagent` tool so it can
* derive and run a child Agent without a reverse dependency on Agent/Session, avoiding circular
* dependencies. The concrete implementation is provided by the SDK composition layer (where
* `createAgent` lives), which internally derives via `createAgent` → `createSession` and hands
* back a `SubagentHandle`.
* Docs: /docs/interfaces § "Subagent interfaces".
*/
export interface SubagentRunner {
/**
* Derives a child Agent and creates a child Session. Precheck errors such as exceeding the
* depth limit or a nonexistent target agent are expressed by throwing (collapsed to `failed`
* by Environment).
*/
spawn(input: {
/** The child Agent's agentId; if omitted, reuses the current Agent (self-invocation). */
agentId?: string;
/** The Model used by the child Session; if omitted, uses the Project's default Model. */
modelId?: string;
}): Promise<SubagentHandle>;
}
/**
* Proxy-reading service for describe_image: injected when the session model doesn't support
* images (vision=false) — images are handed to the configured vision model for description and
* the tool returns text, avoiding a 400 from feeding images back into a tool_result for a
* provider that doesn't support images.
* Docs: /docs/interfaces § "VisionDescriberService".
*/
export interface VisionDescriberService {
/** Vision model id; null when the Project has no `vision_model` configured (or it's invalid), in which case the tool ends with a failed explanation. */
modelId: string | null;
/** Constructs a single-shot LLM for this vision model (no tools, no system prompt); omitted when `modelId` is null. */
createLLM?: () => LLMInterface;
}
/**
* Runtime services Environment injects into individual tools (e.g. `run_subagent` needs `SubagentRunner`); most tools don't use these.
* Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig".
*/
export interface EnvironmentServices {
subagentRunner?: SubagentRunner;
/** Injected when the session model doesn't support images: for describe_image's single-shot vision-model proxy reading. */
visionDescriber?: VisionDescriberService;
/** Registry of long-running command sessions (shared by `exec_command` / `input_command`); constructed and injected internally by Environment. */
commandSessions?: CommandSessionManager;
/** Registry of background subagent sessions (shared by `run_subagent` / `input_subagent`); constructed and injected internally by Environment. */
subagentSessions?: SubagentSessionManager;
}
/** Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig". */
export interface EnvironmentConfig {
workspaceDir: string;
toolConfig: ToolConfig;
/** Runtime services (optional); Environment forwards these to each tool factory to use as needed. */
services?: EnvironmentServices;
/**
* Agent vault environment variables (key-value pairs, taken from the Agent's
* `agent_state/.vault.toml`): injected into the exec_command / input_command subprocess
* environment; hardened entries cannot be overridden.
*/
vault?: Record<string, string>;
}
/**
* An approved tool-call execution request.
* Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig".
*/
export interface ToolExecutionRequest {
/** The OmniMessage whose payload.type === "tool_call". */
toolCall: OmniMessage<ToolCallPayload>;
signal?: AbortSignal;
/** The parent Agent's approval callback; forwarded to tools that need to derive a child Session (run_subagent), implementing approval inheritance. */
approve?: ApproveFn;
}
/**
* Environment interface: executes approved tool calls within the Workspace.
* `executeTool` yields `partial_tool_call_output` as an async generator and ends with exactly one
* complete `tool_call_output`; nested session messages carrying an origin marker (e.g. forwarded
* by run_subagent) pass through unchanged.
*
* **Rendering** of tool calls is not this interface's concern (nor core's): streaming rendering is
* handled by the CLI / Web frontend itself.
* Docs: /docs/interfaces § "EnvironmentInterface".
*/
export interface EnvironmentInterface {
listTools(): Promise<ToolDefinition[]>;
executeTool(request: ToolExecutionRequest): AsyncGenerator<OmniMessage>;
/** Looks up a tool's permission level (for frontend permission-mode decisions); returns undefined for unknown tools. */
toolPermission(name: string): ToolPermission | undefined;
/** Releases runtime resources held by the environment (e.g. managed long-running command sessions); called by the host when the Session ends. Optional, idempotent. */
dispose?(): void;
}
+9
View File
@@ -0,0 +1,9 @@
/** Local-timezone date formatting (internal shared helper, not exported via the barrel). */
/** Format a date as local `yyyy-mm-dd` (local timezone, 4-digit year, zero-padded 2-digit month/day). */
export function formatLocalDate(date: Date): string {
const year = date.getFullYear().toString().padStart(4, "0");
const month = (date.getMonth() + 1).toString().padStart(2, "0");
const day = date.getDate().toString().padStart(2, "0");
return `${year}-${month}-${day}`;
}
@@ -0,0 +1,171 @@
/**
* Session creation helpers (used by `agent.createSession` for assembly, not exported
* via the barrel): Session id generation, runtime environment fields, and temp
* Workspace creation.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { randomBytes, randomUUID } from "node:crypto";
import { formatLocalDate } from "./dates.js";
import type { SessionEnvironmentValues } from "../state/agent-state.js";
import { workspacesDir } from "../state/index.js";
import { userText } from "../omnimessage/index.js";
import type { OmniMessage } from "../omnimessage/index.js";
/** Session runtime environment fields: the placeholder substitution values for `assembleSystemPrompt`; producer and consumer share the same type. */
export type SessionEnvironment = SessionEnvironmentValues;
/** Generate a Session id of the form `session-YYYY-MM-DD-HH-mm-ss-<8-hex>` (local timezone, zero-padded: 4-digit year, 2 digits for the rest; hex from randomUUID). */
export function formatSessionId(date: Date = new Date()): string {
const pad = (n: number) => n.toString().padStart(2, "0");
const ts =
`${formatLocalDate(date)}` +
`-${pad(date.getHours())}-${pad(date.getMinutes())}-${pad(date.getSeconds())}`;
const hex = randomUUID().replace(/-/g, "").slice(0, 8);
return `session-${ts}-${hex}`;
}
/**
* Generate this Session's runtime environment fields (injected via specific
* placeholders in the system prompt).
* This is system-generated runtime context, not sourced from Agent State / Workspace files.
*/
export function sessionEnvironment(
workspaceDir: string,
sessionId: string,
ids: { agentId: string; projectDir: string },
date = new Date(),
): SessionEnvironment {
return {
sessionId,
cwd: workspaceDir,
agentId: ids.agentId,
projectDir: ids.projectDir,
platform: process.platform,
osVersion: getOsVersion(),
date: formatLocalDate(date),
};
}
function getOsVersion(): string {
// os.* is a stable built-in API that normally doesn't throw; but this function only
// builds a single line of environment info for the system prompt, so it's not worth
// letting an exception take down createSession — fall back to "unknown" instead.
try {
if (process.platform === "win32") {
return `${os.version()} ${os.release()}`;
}
return `${os.type()} ${os.release()}`;
} catch {
return "unknown";
}
}
/** The 8-hex space is 2^32, so the odds of consecutive collisions are negligible; the cap only guards against an infinite loop caused by an abnormal filesystem. */
const MAX_TMP_ID_ATTEMPTS = 16;
/**
* Create a temporary Workspace under `<agent>/workspaces/<workspace_id>`, where the
* directory name is the workspace_id, shaped like `tmp-<8hex>`; if it collides with
* an existing directory, regenerate the id. No symlinks are created inside the Workspace:
* the model composes absolute paths (to Agent State, scratchpad, etc.) directly from the
* Environment placeholders (Project Dir / Agent ID) in the system prompt.
*/
export async function createTempWorkspace(
root: string,
projectId: string,
agentId: string,
): Promise<string> {
const base = workspacesDir(root, projectId, agentId);
await fs.mkdir(base, { recursive: true });
// The final directory must use a non-recursive mkdir: recursive mkdir succeeds
// silently when the directory already exists, which would put a new Session into
// an existing temp Workspace; EEXIST means an id collision, so retry with a new id.
for (let attempt = 0; attempt < MAX_TMP_ID_ATTEMPTS; attempt++) {
const dir = path.join(base, `tmp-${randomUUID().slice(0, 8)}`);
try {
await fs.mkdir(dir);
return dir;
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
}
}
throw new Error(
`failed to allocate a unique temp workspace id under ${base} after ${MAX_TMP_ID_ATTEMPTS} attempts`,
);
}
/** Maps a data URL's mime type to a file extension on disk; unknown mimes use bin (the image-reading tool sniffs the magic bytes and doesn't rely on the extension). */
const MIME_TO_EXT: Record<string, string> = {
"image/png": "png",
"image/jpeg": "jpg",
"image/gif": "gif",
"image/webp": "webp",
};
/**
* Input conversion for when the session model doesn't support images: image messages
* in the `run` input are written to disk as files (base64 data URLs are saved to the
* session scratchpad; http(s) URLs are referenced
* as-is), and the path/URL is appended to the user text (an `[attached image: …]`
* line); the image message itself is removed from the input — the model views it by
* path via describe_image (read on its behalf by a vision model), and images never
* enter that session's history directly.
* Returns the input unchanged when there are no images; an image that can't be
* parsed is replaced with an explanatory line rather than silently dropped.
*/
export async function imagesToScratchpadPaths(
input: OmniMessage[],
dir: string,
): Promise<OmniMessage[]> {
const isImage = (m: OmniMessage): boolean =>
(m.payload as { type?: string }).type === "image_url";
if (!input.some(isImage)) return input;
const lines: string[] = [];
for (const msg of input) {
if (!isImage(msg)) continue;
const url = (msg.payload as { image_url?: string }).image_url ?? "";
if (/^https?:\/\//i.test(url)) {
lines.push(`[attached image: ${url}]`);
continue;
}
const match = /^data:([^;,]+);base64,(.+)$/s.exec(url);
if (!match) {
lines.push("[an attached image could not be saved and was dropped]");
continue;
}
await fs.mkdir(dir, { recursive: true });
const ext = MIME_TO_EXT[match[1]!.toLowerCase()] ?? "bin";
// Filename = upload-<8 random hex chars> (same convention as project-<8hex>; the
// prefix distinguishes model-generated temp files).
// "wx" flag does exclusive creation to avoid name collisions: on the rare chance of a collision, retry with a new random value.
let file: string;
for (;;) {
file = path.join(dir, `upload-${randomBytes(4).toString("hex")}.${ext}`);
try {
await fs.writeFile(file, Buffer.from(match[2]!, "base64"), { flag: "wx" });
break;
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
}
}
lines.push(`[attached image: ${file}]`);
}
// Concatenation: the path lines are appended after the last user text message; if the input is images only, add a plain path-only text message.
const rest = input.filter((m) => !isImage(m));
const suffix = lines.join("\n");
const lastTextIdx = rest.findLastIndex((m) => {
const p = m.payload as { type?: string; role?: string };
return p.type === "text" && p.role === "user";
});
if (lastTextIdx === -1) return [...rest, userText(suffix)];
return rest.map((m, i) => {
if (i !== lastTextIdx) return m;
const p = m.payload as { type: string; role: string; text: string };
return { ...m, payload: { ...p, text: `${p.text}\n\n${suffix}` } } as OmniMessage;
});
}
File diff suppressed because it is too large Load Diff
+22
View File
@@ -0,0 +1,22 @@
/**
* LLM interface module entry point.
*
* Exports `GenerativeModel` (the LLMInterface implementation), along with internal
* pure conversion functions for unit testing (message merging, event translation,
* token accounting, UniConfig construction, retry determination).
*/
export {
GenerativeModel,
EventTranslator,
groupHistoryToUniMessages,
mergeOmniToUniMessage,
translateEvents,
usageToTokenCounts,
isMalformedJsonParseError,
isIncompleteStreamError,
isRetryableError,
mapThinkingLevel,
toolDefinitionsToSchemas,
buildUniConfig,
} from "./generative-model.js";
export { ToolCallIdAllocator, stripToolCallIdSuffix } from "./tool-call-ids.js";
+51
View File
@@ -0,0 +1,51 @@
/**
* Session-level uniqueness for tool_call_id.
*
* Some providers don't produce a real call id: e.g. Gemini's functionCall has no id, so AgentHub
* uses the **function name** as the `tool_call_id` — consecutive/parallel calls to the same tool then
* all share one id. But the OmniMessage world (engine dispatch/pairing, approval routing, frontend
* tool-card attribution) keys on `tool_call_id`, and a collision lets a later call overwrite the
* earlier one (parallel same-name calls in one turn can even be dropped entirely).
*
* Approach: inbound, `EventTranslator` disambiguates duplicate ids with a `#n` suffix (the first keeps
* the original id); outbound (returning tool_result, replaying history on resume) uses
* `stripToolCallIdSuffix` to strip the suffix and restore the original — Gemini's functionResponse
* pairs by using `tool_call_id` as the name, so it must be restored to the function name. The registry
* lives at Session level (the new GenerativeModel rebuilt on compaction shares the same instance), and
* on resume `setHistory` seeds it with historical ids, so the uniqueness scope covers the entire
* context the frontend renders.
* Docs: /docs/interfaces § "The built-in implementation: GenerativeModel".
*/
export class ToolCallIdAllocator {
/** OmniMessage-level tool_call_ids already taken in this Session (history-seeded + allocated). */
private used = new Set<string>();
/** Register an already-used id (for resume seeding); registering twice is harmless. */
markUsed(id: string): void {
this.used.add(id);
}
/**
* Allocate a Session-unique id for a provider-reported tool_call_id: if unused, keep the original;
* if already used (a repeat call from a name-as-id provider), take the first free `origId#n` (n from 2).
* Providers with truly unique ids (OpenAI `call_*` / Claude `toolu_*`) never collide, so they pass through unchanged.
*/
allocate(providerId: string): string {
let id = providerId;
for (let n = 2; this.used.has(id); n += 1) {
id = `${providerId}#${n}`;
}
this.used.add(id);
return id;
}
}
/**
* Strip the `#n` suffix added by `allocate`, restoring the provider's original id (returns as-is when
* there's no suffix; idempotent). On resume there's no registry to compare against, so it trims by
* shape: real ids from known providers (OpenAI/Claude `call_*`/`toolu_*`, Gemini function names — `#`
* isn't a valid function-name char) never end in `#<digits>`, so they aren't harmed.
*/
export function stripToolCallIdSuffix(id: string): string {
return id.replace(/#\d+$/, "");
}
+155
View File
@@ -0,0 +1,155 @@
/**
* Aggregates streaming partial_* messages into a complete model_msg.
*
* When recording a Trace, streaming `partial_*` messages must first be joined into a complete
* `model_msg` before writing. This module provides:
* - `PartialAggregator`: a stateful aggregator, pushed one message at a time, producing a
* complete message when a fragment ends with `stop`;
* - `aggregateAll`: a one-shot pass that collapses `partial_*` messages in an array into
* complete messages.
*
* Complete / event / session_meta messages pass through unchanged, preserving their original
* order.
* Docs: /docs/omni-message § "The streaming discipline".
*/
import { assistantText, thinkingMessage, toolCall, toolCallOutput } from "./builders.js";
import type { OmniMessage, PartialModelPayload, StopReason } from "./types.js";
import { isPartialPayload } from "./types.js";
type PartialKind = PartialModelPayload["type"];
interface OpenFragment {
kind: PartialKind;
/** Accumulation buffer for text / thinking / tool_call arguments / tool_call_output. */
buffer: string;
name?: string;
toolCallId?: string;
/** Images carried by tool_call_output (images aren't incremental — a single delta carries the whole set; a later one overwrites). */
images?: string[];
lastStopReason: StopReason;
}
/** Merge key for partial fragments: same type + same tool_call_id counts as the same fragment. */
function fragmentKey(p: PartialModelPayload): string {
const id = "tool_call_id" in p ? p.tool_call_id : "";
return `${p.type}::${id}`;
}
function finalize(frag: OpenFragment): OmniMessage {
switch (frag.kind) {
case "partial_text":
return assistantText(frag.buffer, frag.lastStopReason);
case "partial_thinking":
return thinkingMessage(frag.buffer, frag.lastStopReason);
case "partial_tool_call":
return toolCall({
name: frag.name ?? "",
arguments: frag.buffer,
toolCallId: frag.toolCallId ?? "",
stopReason: frag.lastStopReason,
});
case "partial_tool_call_output":
return toolCallOutput({
output: frag.buffer,
toolCallId: frag.toolCallId ?? "",
stopReason: frag.lastStopReason,
...(frag.images !== undefined ? { images: frag.images } : {}),
});
}
}
function appendDelta(frag: OpenFragment, p: PartialModelPayload): void {
switch (p.type) {
case "partial_text":
frag.buffer += p.text;
break;
case "partial_thinking":
frag.buffer += p.thinking;
break;
case "partial_tool_call":
frag.buffer += p.arguments;
if (p.name) frag.name = p.name;
frag.toolCallId = p.tool_call_id;
break;
case "partial_tool_call_output":
frag.buffer += p.output;
if (p.images && p.images.length > 0) frag.images = p.images;
frag.toolCallId = p.tool_call_id;
break;
}
if (p.stop_reason !== undefined) frag.lastStopReason = p.stop_reason;
}
function newFragment(p: PartialModelPayload): OpenFragment {
const frag: OpenFragment = {
kind: p.type,
buffer: "",
lastStopReason: "completed",
};
if (p.type === "partial_tool_call") {
frag.name = p.name;
frag.toolCallId = p.tool_call_id;
} else if (p.type === "partial_tool_call_output") {
frag.toolCallId = p.tool_call_id;
}
return frag;
}
/**
* Stateful aggregator. Pushed one message at a time via `push`:
* - complete / event / session_meta messages are returned unchanged;
* - `partial_*` messages accumulate into an internal fragment, producing a complete message
* when `event_type === "stop"`;
* - `flush` forcibly emits any fragments that haven't yet received a stop.
*/
export class PartialAggregator {
private open = new Map<string, OpenFragment>();
push(msg: OmniMessage): OmniMessage[] {
if (!isPartialPayload(msg.payload)) {
return [msg];
}
const p = msg.payload;
const key = fragmentKey(p);
let frag = this.open.get(key);
if (p.event_type === "start") {
// start reopens a fragment; if a fragment with the same key already exists (out-of-order), finalize it first.
const out: OmniMessage[] = [];
if (frag) out.push(finalize(frag));
frag = newFragment(p);
appendDelta(frag, p);
this.open.set(key, frag);
return out;
}
if (!frag) {
// delta/stop without a preceding start: handle leniently, creating a new fragment as needed.
frag = newFragment(p);
this.open.set(key, frag);
}
appendDelta(frag, p);
if (p.event_type === "stop") {
this.open.delete(key);
return [finalize(frag)];
}
return [];
}
/** Finalizes: emits all still-open fragments (in the order they were opened). */
flush(): OmniMessage[] {
const out = [...this.open.values()].map(finalize);
this.open.clear();
return out;
}
}
/** One-shot aggregation: keeps non-partial messages in their original order, collapsing partial ones into complete messages. */
export function aggregateAll(messages: OmniMessage[]): OmniMessage[] {
const agg = new PartialAggregator();
const out: OmniMessage[] = [];
for (const msg of messages) out.push(...agg.push(msg));
out.push(...agg.flush());
return out;
}
+342
View File
@@ -0,0 +1,342 @@
/**
* OmniMessage builders. All modules create messages exclusively through these builders, avoiding
* ad hoc protocol structures scattered across the codebase.
* Every builder writes an ISO 8601 UTC timestamp.
* Docs: /docs/omni-message § "Builders and guards".
*/
import type {
AbortPayload,
ApprovalDecision,
ApprovalDecisionPayload,
CompactionBeginPayload,
CompactionEndPayload,
CompactionMode,
CompactionReason,
EventMessage,
ImageUrlPayload,
InlineDataPayload,
InlineThinkingPayload,
MessageOrigin,
ModelMessage,
OmniMessage,
PartialTextPayload,
PartialThinkingPayload,
PartialToolCallOutputPayload,
PartialToolCallPayload,
RequestBeginPayload,
RequestEndPayload,
Role,
SessionMetaMessage,
SessionMetaPayload,
StopReason,
StreamEventType,
SubagentPayload,
TextPayload,
ThinkingPayload,
TokenCounts,
TokenUsagePayload,
ToolCallOutputPayload,
ToolCallPayload,
} from "./types.js";
/** The current moment's ISO 8601 UTC timestamp. */
function nowIso(): string {
return new Date().toISOString();
}
function model<P extends ModelMessage["payload"]>(payload: P): OmniMessage<P> {
return { timestamp: nowIso(), type: "model_msg", payload };
}
function event<P extends EventMessage["payload"]>(payload: P): OmniMessage<P> {
return { timestamp: nowIso(), type: "event_msg", payload };
}
// session_meta ---------------------------------------------------------------
export function sessionMeta(payload: SessionMetaPayload): SessionMetaMessage {
return { timestamp: nowIso(), type: "session_meta", payload };
}
// Complete model_msg -----------------------------------------------------------
/**
* Provider fidelity fields: kept as-is and restored verbatim on replay.
* Builder convention: positional-argument-style builders carry these in a trailing `fidelity`
* object (narrowed via Pick per payload type — e.g. thinking only has signature); object-argument-
* style builders (toolCall) flatten `fidelity` fields into the parameter object alongside
* `stopReason`, mirroring the payload structure directly.
*/
export interface FidelityFields {
phase?: string | null;
signature?: string;
}
export function textMessage(
role: Role,
text: string,
stopReason: StopReason = "completed",
fidelity?: FidelityFields,
): OmniMessage<TextPayload> {
return model({
type: "text",
role,
text,
stop_reason: stopReason,
...(fidelity?.phase != null ? { phase: fidelity.phase } : {}),
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export const userText = (text: string): OmniMessage<TextPayload> => textMessage("user", text);
export const assistantText = (
text: string,
stopReason: StopReason = "completed",
fidelity?: FidelityFields,
): OmniMessage<TextPayload> => textMessage("assistant", text, stopReason, fidelity);
export function imageUrlMessage(imageUrl: string): OmniMessage<ImageUrlPayload> {
return model({
type: "image_url",
role: "user",
image_url: imageUrl,
stop_reason: "completed",
});
}
export function inlineData(
role: Role,
data: string,
mimeType: string,
fidelity?: Pick<FidelityFields, "signature">,
): OmniMessage<InlineDataPayload> {
return model({
type: "inline_data",
role,
data,
mime_type: mimeType,
stop_reason: "completed",
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export function thinkingMessage(
thinking: string,
stopReason: StopReason = "completed",
fidelity?: Pick<FidelityFields, "signature">,
): OmniMessage<ThinkingPayload> {
return model({
type: "thinking",
role: "assistant",
thinking,
stop_reason: stopReason,
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export function inlineThinking(
data: string,
mimeType: string,
fidelity?: Pick<FidelityFields, "signature">,
): OmniMessage<InlineThinkingPayload> {
return model({
type: "inline_thinking",
role: "assistant",
data,
mime_type: mimeType,
stop_reason: "completed",
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export function toolCall(args: {
name: string;
arguments: string;
toolCallId: string;
stopReason?: StopReason;
signature?: string;
}): OmniMessage<ToolCallPayload> {
return model({
type: "tool_call",
role: "assistant",
name: args.name,
arguments: args.arguments,
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
...(args.signature !== undefined ? { signature: args.signature } : {}),
});
}
export function toolCallOutput(args: {
output: string;
toolCallId: string;
stopReason?: StopReason;
/** Images carried by the tool output (array of data URLs); images aren't incremental — a single delta carries the whole set in the streaming path, and the complete message carries them too. */
images?: string[];
}): OmniMessage<ToolCallOutputPayload> {
return model({
type: "tool_call_output",
role: "user",
output: args.output,
...(args.images !== undefined && args.images.length > 0 ? { images: args.images } : {}),
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
});
}
// Streaming partial_* model_msg -------------------------------------------------
export function partialText(
eventType: StreamEventType,
text = "",
stopReason: StopReason = "completed",
): OmniMessage<PartialTextPayload> {
return model({
type: "partial_text",
role: "assistant",
event_type: eventType,
text,
stop_reason: stopReason,
});
}
export function partialThinking(
eventType: StreamEventType,
thinking = "",
stopReason: StopReason = "completed",
): OmniMessage<PartialThinkingPayload> {
return model({
type: "partial_thinking",
role: "assistant",
event_type: eventType,
thinking,
stop_reason: stopReason,
});
}
export function partialToolCall(args: {
eventType: StreamEventType;
name: string;
arguments?: string;
toolCallId: string;
stopReason?: StopReason;
}): OmniMessage<PartialToolCallPayload> {
return model({
type: "partial_tool_call",
role: "assistant",
event_type: args.eventType,
name: args.name,
arguments: args.arguments ?? "",
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
});
}
export function partialToolCallOutput(args: {
eventType: StreamEventType;
output?: string;
toolCallId: string;
stopReason?: StopReason;
/** Images carried by the tool output (array of data URLs); images aren't incremental — a single delta carries the whole set. */
images?: string[];
}): OmniMessage<PartialToolCallOutputPayload> {
return model({
type: "partial_tool_call_output",
role: "user",
event_type: args.eventType,
output: args.output ?? "",
...(args.images !== undefined && args.images.length > 0 ? { images: args.images } : {}),
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
});
}
// event_msg -------------------------------------------------------------------
export function approvalDecision(
decision: ApprovalDecision,
toolCallId: string,
): OmniMessage<ApprovalDecisionPayload> {
return event({ type: "approval_decision", decision, tool_call_id: toolCallId });
}
export function abortEvent(reason: string | null = null): OmniMessage<AbortPayload> {
return event({ type: "abort", reason });
}
/** request begin event: marks the start of one LLM Request. */
export function requestBegin(): OmniMessage<RequestBeginPayload> {
return event({ type: "request_begin" });
}
/** request end event: carries the terminal state (`completed` means this turn was already committed to AgentHub). */
export function requestEnd(status: StopReason): OmniMessage<RequestEndPayload> {
return event({ type: "request_end", status });
}
/** compaction begin event: carries the trigger reason, mode, current context usage, and cumulative Session turn count. */
export function compactionBegin(args: {
reason: CompactionReason;
mode: CompactionMode;
context: number;
turns: number;
}): OmniMessage<CompactionBeginPayload> {
return event({
type: "compaction_begin",
reason: args.reason,
mode: args.mode,
context: args.context,
turns: args.turns,
});
}
/** compaction end event: carries the compaction result (non-`completed` means compaction was abandoned and the original context is kept). */
export function compactionEnd(args: {
reason: CompactionReason;
mode: CompactionMode;
status: StopReason;
}): OmniMessage<CompactionEndPayload> {
return event({
type: "compaction_end",
reason: args.reason,
mode: args.mode,
status: args.status,
});
}
/** subagent derivation pointer event: records only the direct child session's Session id (written to the parent Trace by context_engine). */
export function subagentEvent(sessionId: string): OmniMessage<SubagentPayload> {
return event({ type: "subagent", session_id: sessionId });
}
export function emptyTokenCounts(): TokenCounts {
return { cache_read: 0, cache_write: 0, output: 0, total: 0 };
}
export function tokenUsage(
session: TokenCounts,
request: TokenCounts,
): OmniMessage<TokenUsagePayload> {
return event({ type: "token_usage", session, request });
}
/** Adds two sets of Token counts together, used to maintain cumulative Session usage. */
export function addTokenCounts(a: TokenCounts, b: TokenCounts): TokenCounts {
return {
cache_read: a.cache_read + b.cache_read,
cache_write: a.cache_write + b.cache_write,
output: a.output + b.output,
total: a.total + b.total,
};
}
/**
* Marks a message with a nested-origin tag: prepends one hop (a child Session id) to the front
* of `origin`, outer-to-inner.
* Used by host tools (e.g. run_subagent) when forwarding child-session messages; an absent
* `origin` means the message comes from the main Session.
*/
export function withOrigin<M extends OmniMessage>(msg: M, sessionId: MessageOrigin): M {
return { ...msg, origin: [sessionId, ...(msg.origin ?? [])] };
}
+3
View File
@@ -0,0 +1,3 @@
export * from "./types.js";
export * from "./builders.js";
export * from "./aggregate.js";
+400
View File
@@ -0,0 +1,400 @@
/**
* OmniMessage — PenguinHarness's primary message protocol.
*
* All messages share one envelope: `timestamp` (ISO 8601 UTC), `type`, and `payload`.
* The outer `type` falls into three categories:
* - `session_meta`: Session metadata;
* - `model_msg`: model input/output messages (both complete messages and streaming
* `partial_*` messages);
* - `event_msg`: control/statistics events during execution.
*
* Trace records only: `session_meta`, complete `model_msg`, and all `event_msg`;
* the Human interface communicates using: complete `model_msg`, streaming `partial_*`, and all
* `event_msg`.
*
* Docs: packages/docs/content/omni-message.{zh,en}.md (site path /docs/omni-message) documents
* this protocol payload-for-payload — keep the page in sync when changing types here.
*/
/** The outer message category. */
export type OmniMessageType = "session_meta" | "model_msg" | "event_msg";
/** The message's originating role. */
export type Role = "user" | "assistant";
/**
* The reason a model response or message generation ended. Only five protocol values are
* allowed:
* - `completed`: finished normally, including completed text, thinking, tool requests, or
* tool output;
* - `failed`: a non-retryable error or tool execution failure;
* - `aborted`: user-initiated interruption or cancellation;
* - `timeout`: LLM request timed out;
* - `malformed`: the LLM response was malformed (e.g. AgentHub JSON parsing exception).
* Only LLM timeout / malformed trigger a context_engine reconnect.
* Docs: /docs/omni-message § "stop_reason".
*/
export type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed";
/** The event phase of a streaming fragment. `stop` marks the end of a fragment and usually carries no incremental content. */
export type StreamEventType = "start" | "delta" | "stop";
/**
* Nested-origin marker: a child Session id. The message envelope's `origin` is a chain of child
* Session ids ordered **outer-to-inner**, identifying that the message comes from a nested child
* session (e.g. a child Session derived by `run_subagent`); each layer of host-tool forwarding
* prepends one more hop at the front. **An absent `origin` (the message carries no `origin`)
* means the message comes from the main Session itself** (an empty array is never produced
* either). Only session_id is recorded: the corresponding tool_call / agent info can be obtained
* from the `run_subagent` tool_call in the parent session's stream and the child Session's own
* Trace (session_meta).
* Docs: /docs/omni-message § "origin: the Subagent chain".
*/
export type MessageOrigin = string;
/** The approval decision for a tool call. */
export type ApprovalDecision = "allow" | "deny";
/** Token counts (input/output/cache/total). */
export interface TokenCounts {
cache_read: number;
cache_write: number;
output: number;
total: number;
}
// ---------------------------------------------------------------------------
// session_meta
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "session_meta"
/** Tool definition passed to the LLM (OpenAI/JSON Schema style). */
export interface ToolDefinition {
name: string;
description: string;
parameters?: Record<string, unknown>;
}
export interface SessionMetaPayload {
session_id: string;
/** The session model's provider group (paired with `model_id` to form a model reference). */
provider: string;
/** The session model's upstream model_id (the request id sent to AgentHub; paired with `provider`). */
model_id: string;
model_context_window: number | string;
/** The system prompt actually used by this Session (the assembled result with environment placeholders already substituted). */
system_prompt: string;
/** The list of tool definitions this Session exposes to the model (full schema, matching what's sent to the LLM). */
tools: ToolDefinition[];
/** The model's thinking level (from system_config.model.thinking_level; "default" when unconfigured). */
thinking_level: string;
/** Absolute path to the Agent State. */
agent_state: string;
/** Absolute path to the Workspace. */
workspace: string;
}
// ---------------------------------------------------------------------------
// model_msg — complete messages
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "model_msg: complete payloads"
export interface TextPayload {
type: "text";
role: Role;
text: string;
stop_reason?: StopReason;
/** Provider fidelity field: text phase marker (e.g. GPT-5 segments by phase), kept as-is and restored verbatim. */
phase?: string | null;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ImageUrlPayload {
type: "image_url";
role: "user";
/** A web URL or a base64 data URL. */
image_url: string;
stop_reason?: StopReason;
}
export interface InlineDataPayload {
type: "inline_data";
role: Role;
/** Base64-encoded bytes. */
data: string;
mime_type: string;
stop_reason?: StopReason;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ThinkingPayload {
type: "thinking";
role: "assistant";
thinking: string;
stop_reason?: StopReason;
/**
* Provider fidelity field: thinking-block signature (Claude thinking blocks / redacted
* thinking, GPT-5 encrypted reasoning, etc. — **required** when some models replay history),
* kept as-is and restored verbatim — losing it breaks Session resumption.
*/
signature?: string;
}
export interface InlineThinkingPayload {
type: "inline_thinking";
role: "assistant";
/** Base64-encoded bytes. */
data: string;
mime_type: string;
stop_reason?: StopReason;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ToolCallPayload {
type: "tool_call";
role: "assistant";
name: string;
/** Tool arguments as a JSON string. */
arguments: string;
tool_call_id: string;
stop_reason?: StopReason;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ToolCallOutputPayload {
type: "tool_call_output";
role: "user";
output: string;
/**
* Images carried by the tool output (optional): each is a `data:<mime>;base64,...` data URL,
* fed back to the model alongside the text (e.g. images read by read_image). Images aren't
* incremental: the streaming path carries the whole set once via a single delta (see
* `PartialToolCallOutputPayload.images`), and the complete message carries them again — the
* streamed-and-joined result equals the complete message.
*/
images?: string[];
tool_call_id: string;
stop_reason?: StopReason;
}
// ---------------------------------------------------------------------------
// model_msg — streaming partial_* messages
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "model_msg: streaming partials"
export interface PartialTextPayload {
type: "partial_text";
role: "assistant";
event_type: StreamEventType;
text: string;
stop_reason?: StopReason;
}
export interface PartialThinkingPayload {
type: "partial_thinking";
role: "assistant";
event_type: StreamEventType;
thinking: string;
stop_reason?: StopReason;
}
export interface PartialToolCallPayload {
type: "partial_tool_call";
role: "assistant";
event_type: StreamEventType;
name: string;
/** Incremental fragment of the arguments JSON. */
arguments: string;
tool_call_id: string;
stop_reason?: StopReason;
}
export interface PartialToolCallOutputPayload {
type: "partial_tool_call_output";
role: "user";
event_type: StreamEventType;
output: string;
/** Images carried by the tool output (optional): images aren't incremental, carried as a whole by a single delta (consistent with the complete message). */
images?: string[];
tool_call_id: string;
stop_reason?: StopReason;
}
// ---------------------------------------------------------------------------
// event_msg
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "event_msg"
export interface ApprovalDecisionPayload {
type: "approval_decision";
decision: ApprovalDecision;
tool_call_id: string;
}
export interface AbortPayload {
type: "abort";
reason?: string | null;
}
export interface TokenUsagePayload {
type: "token_usage";
/** Current Session cumulative token usage. */
session: TokenCounts;
/** Token usage for the most recent Request. */
request: TokenCounts;
}
/**
* Request boundary event: the boundary of one LLM Request, produced **in pairs** by
* `context_engine` and written to Trace. `request_end`
* with `status` of `completed` means the turn has been committed by AgentHub — this is the
* mechanical criterion Trace replay (Session resumption) uses to determine whether a turn was
* committed, and it also gives performance analysis a basis for Request latency and turn counts.
* A compaction request produces this same event pair too (written to Trace only, not streamed).
*/
export interface RequestBeginPayload {
type: "request_begin";
}
export interface RequestEndPayload {
type: "request_end";
/** Terminal state of this Request (reuses the five StopReason values, sharing its source with this turn's complete message's stop_reason / LLMOutcome). */
status: StopReason;
}
/** Compaction trigger reason: context threshold / turn-count threshold / user-initiated request. */
export type CompactionReason = "context" | "turns" | "manual";
/** Context compaction mode: summary relay / direct discard. */
export type CompactionMode = "summarize" | "discard";
/**
* Compaction boundary event: the compaction process exposes
* only this event pair to Human, produced **in pairs** by `context_engine`. Both `reason` and
* `mode` are carried on both events, for stateless frontend rendering; `status` reuses the
* five-value `StopReason` protocol (compaction converges to a terminal state, taking
* `completed` / `failed` / `aborted` in practice — `timeout` / `malformed` are handled internally
* by the compaction request's existing retry mechanism, collapsing to `failed` once retries are
* exhausted).
*/
export interface CompactionBeginPayload {
type: "compaction_begin";
reason: CompactionReason;
mode: CompactionMode;
/** Current context token usage (the most recent token_usage's request.total). */
context: number;
/** Session cumulative turn count. */
turns: number;
}
export interface CompactionEndPayload {
type: "compaction_end";
reason: CompactionReason;
mode: CompactionMode;
/** Compaction result; non-`completed` means compaction was abandoned and the original context was kept. */
status: StopReason;
}
/**
* Subagent pointer event: when the parent Session spawns a
* **direct** child session, `context_engine` writes this to the parent Trace (not streamed),
* recording only the child session's Session id — the child session's other details live in its
* own Trace's `session_meta`. When the session is reopened, the server uses this to recursively
* expand the child Trace and reconstruct the `origin` chain; a grandchild session's pointer is
* recorded by the child Trace itself.
*/
export interface SubagentPayload {
type: "subagent";
/** The direct child session's Session id. */
session_id: string;
}
// ---------------------------------------------------------------------------
// Union types and the message envelope
// ---------------------------------------------------------------------------
/** Complete model_msg payload (written to Trace and exposed externally). */
export type CompleteModelPayload =
| TextPayload
| ImageUrlPayload
| InlineDataPayload
| ThinkingPayload
| InlineThinkingPayload
| ToolCallPayload
| ToolCallOutputPayload;
/** Streaming model_msg payload. */
export type PartialModelPayload =
| PartialTextPayload
| PartialThinkingPayload
| PartialToolCallPayload
| PartialToolCallOutputPayload;
export type ModelPayload = CompleteModelPayload | PartialModelPayload;
export type EventPayload =
| ApprovalDecisionPayload
| AbortPayload
| RequestBeginPayload
| RequestEndPayload
| TokenUsagePayload
| CompactionBeginPayload
| CompactionEndPayload
| SubagentPayload;
export type OmniPayload = SessionMetaPayload | ModelPayload | EventPayload;
/** The unified message envelope. */
export interface OmniMessage<P extends OmniPayload = OmniPayload> {
/** ISO 8601 UTC timestamp. */
timestamp: string;
type: OmniMessageType;
payload: P;
/** Nested-origin marker: the chain of child Session ids ordered outer-to-inner; absent = from the main Session (see MessageOrigin). */
origin?: MessageOrigin[];
}
// Convenience aliases for concrete message types --------------------------------
export type SessionMetaMessage = OmniMessage<SessionMetaPayload>;
export type ModelMessage = OmniMessage<ModelPayload>;
export type EventMessage = OmniMessage<EventPayload>;
export type CompleteModelMessage = OmniMessage<CompleteModelPayload>;
export type PartialModelMessage = OmniMessage<PartialModelPayload>;
// ---------------------------------------------------------------------------
// Runtime discrimination helpers
// ---------------------------------------------------------------------------
/** The set of type values for streaming partial_* payloads. */
const PARTIAL_PAYLOAD_TYPES = [
"partial_text",
"partial_thinking",
"partial_tool_call",
"partial_tool_call_output",
] as const;
export function isPartialPayload(p: OmniPayload): p is PartialModelPayload {
return (PARTIAL_PAYLOAD_TYPES as readonly string[]).includes((p as { type?: string }).type ?? "");
}
export function isModelMessage(msg: OmniMessage): msg is ModelMessage {
return msg.type === "model_msg";
}
export function isEventMessage(msg: OmniMessage): msg is EventMessage {
return msg.type === "event_msg";
}
export function isSessionMeta(msg: OmniMessage): msg is SessionMetaMessage {
return msg.type === "session_meta";
}
/** A complete model_msg (not partial_*), i.e. a message that can be written to Trace. */
export function isCompleteModelMessage(msg: OmniMessage): msg is CompleteModelMessage {
return msg.type === "model_msg" && !isPartialPayload(msg.payload);
}
+124
View File
@@ -0,0 +1,124 @@
/**
* Session title generation: an **out-of-band, one-off request** that generates
* a short title from the first-turn conversation text.
*
* Called by `session.generateTitle()`: sends one request using the bare LLM for the session's
* Model (no tools, no system prompt, thinking off), without writing history or Trace. Material
* defaults to what the Session self-captures during run (see session.ts); this module is only
* responsible for the prompt format, driving the one-off request, and sanitizing the result —
* when to generate a title and where to store it is decided by the host (Web server / CLI).
*/
import { userText } from "./omnimessage/index.js";
import type {
OmniMessage,
TextPayload,
TokenCounts,
TokenUsagePayload,
} from "./omnimessage/index.js";
import type { LLMInterface } from "./interfaces.js";
/** Cap on conversation text spliced into the title request (user/model each truncated separately, to control cost). */
const EXCERPT_MAX_CHARS = 2000;
/** Cap on title length (fallback truncation for when the model occasionally ignores the constraint). */
const TITLE_MAX_CHARS = 30;
export interface SessionTitleResult {
/** The sanitized title; null when material is insufficient, the request fails, or the output is empty. */
title: string | null;
/** Token consumption for this request (accumulated token_usage.request); null if no request occurred or there's no usage. */
usage: TokenCounts | null;
}
/**
* Assembles the title-generation Prompt (exported for host/test assertion use). Uses English
* instructions to avoid polluting the title's language, and requires the **title to be in the
* same language as the conversation** (English conversation gets an English title, Chinese
* conversation gets a Chinese title); when assistant material is empty, it relies on the user
* request alone.
*/
export function buildTitlePrompt(userExcerpt: string, assistantExcerpt: string): string {
const clip = (s: string) => (s.length > EXCERPT_MAX_CHARS ? s.slice(0, EXCERPT_MAX_CHARS) : s);
const lines = [
"Generate a concise title for the conversation below.",
"Rules:",
"- Write the title in the SAME language the user is using.",
"- Keep it short: at most 6 words, or ~16 characters for CJK.",
"- Output ONLY the title text — no quotes, no trailing punctuation, no explanation.",
"",
"[User]",
clip(userExcerpt),
];
if (assistantExcerpt.trim()) {
lines.push("", "[Assistant]", clip(assistantExcerpt));
}
return lines.join("\n");
}
/** Sanitizes model output into a title: strips leading/trailing quotes/brackets and trailing punctuation (until stable), collapses whitespace, and truncates if too long; returns null for an empty result. */
export function sanitizeTitle(raw: string): string | null {
let t = raw.replace(/\s+/g, " ").trim();
// Stripping quotes can expose more punctuation underneath (or vice versa), so strip repeatedly until stable.
for (let prev = ""; prev !== t;) {
prev = t;
t = t
.replace(/^["'“”‘’「」『』《》〈〉【】()()\s]+/, "")
.replace(/["'“”‘’「」『』《》〈〉【】()()\s]+$/, "")
.replace(/[。..!!??;;,,、::]+$/, "")
.trim();
}
if (!t) return null;
return t.length > TITLE_MAX_CHARS ? t.slice(0, TITLE_MAX_CHARS) : t;
}
/**
* Drives a single title-generation request: collects model text and token_usage, and resolves
* based on the outcome. Generation only requires user material (assistant material may be
* empty — a pure tool-only turn can still get a title); no request is sent if user material is
* empty; `title` is null if the request doesn't complete (any usage already produced is still
* returned).
*/
export async function generateTitleWithLLM(
llm: LLMInterface,
args: { userText: string; assistantText: string; signal?: AbortSignal },
): Promise<SessionTitleResult> {
if (!args.userText.trim()) {
return { title: null, usage: null };
}
const prompt = buildTitlePrompt(args.userText, args.assistantText);
const gen = llm.streamGenerate({
newMessages: [userText(prompt)],
...(args.signal ? { signal: args.signal } : {}),
});
let collected = "";
let usage: TokenCounts | null = null;
for (;;) {
const step = await gen.next();
if (step.done) {
if (step.value.status !== "completed") return { title: null, usage };
break;
}
const msg = step.value;
if (isAssistantText(msg)) collected += msg.payload.text;
if (isTokenUsage(msg)) {
const r = msg.payload.request;
usage = usage
? {
cache_read: usage.cache_read + r.cache_read,
cache_write: usage.cache_write + r.cache_write,
output: usage.output + r.output,
total: usage.total + r.total,
}
: { ...r };
}
}
return { title: sanitizeTitle(collected), usage };
}
function isAssistantText(msg: OmniMessage): msg is OmniMessage<TextPayload> {
const payload = msg.payload as { type?: string; role?: string };
return msg.type === "model_msg" && payload.type === "text" && payload.role === "assistant";
}
function isTokenUsage(msg: OmniMessage): msg is OmniMessage<TokenUsagePayload> {
return msg.type === "event_msg" && (msg.payload as { type?: string }).type === "token_usage";
}
+250
View File
@@ -0,0 +1,250 @@
/**
* Session — a continuous conversation context under the same Agent and Workspace.
*
* Human is the SDK's input/output boundary: there is no "Human
* implementation/interface".
* - Input: the OmniMessage list (Prompt) passed to `run(newMessages, opts?)`, plus the abort
* signal `signal` and the per-call approval callback `approve` in `opts`;
* - Output: `run` streams OmniMessage via an async generator.
*
* Approval is a **within-turn interaction**: as soon as a tool_call finishes streaming, `approve`
* is requested immediately, and it executes if allowed. Approvals for multiple tools happen one
* at a time, but execution doesn't block the generation/approval of subsequent tools (execution
* can overlap). GenerativeModel maintains history across turns/Tasks. A Task ends when a turn no
* longer produces a tool_call (final reply).
*
* Rendering tool calls is not Session/core's responsibility: the CLI / Web frontend renders it
* from the streamed OmniMessage on its own.
* Docs: /docs/agent-loop; /docs/interfaces § "The Human boundary".
*/
import { sessionMeta } from "./omnimessage/index.js";
import type { OmniMessage, SessionMetaPayload, TokenCounts } from "./omnimessage/index.js";
import { imagesToScratchpadPaths } from "./internal/session-support.js";
import type { EnvironmentInterface, LLMInterface, ToolPermission } from "./interfaces.js";
import { generateTitleWithLLM } from "./session-title.js";
import type { SessionTitleResult } from "./session-title.js";
import { ContextEngine } from "./engine/context-engine.js";
import type {
CompactAvailability,
CompactionSettings,
EngineInitialState,
RunOptions,
TraceSink,
} from "./engine/context-engine.js";
export interface SessionConfig {
/** Session metadata (session_id / provider / model_id / model_context_window / system_prompt / tools / thinking_level / agent_state / workspace). */
meta: SessionMetaPayload;
llm: LLMInterface;
environment: EnvironmentInterface;
trace?: TraceSink;
maxTurns?: number;
/** Creates a new LLM object after compaction (carries over the Session's accumulated Token count); context compaction is unavailable if not provided. */
createLLM?: (sessionTokens: TokenCounts) => LLMInterface;
/**
* Factory for the bare LLM used by out-of-band, one-off requests (same Model/credential as
* the session; no tools, no system prompt, thinking off): used for meta-requests such as
* `generateTitle`; if not provided, `generateTitle` returns null.
*/
createBareLLM?: () => LLMInterface;
/** Context compaction settings (defaults are filled in by the composition layer); only takes effect when provided together with `createLLM`. */
compaction?: CompactionSettings;
/** Session resume: `session_meta` is already in the original Trace file, so it isn't written again on the first run (avoids duplication). */
metaAlreadyWritten?: boolean;
/** Session resume: the engine's initial state derived from Trace replay (carry-over / accumulated stats, etc.). */
initialEngineState?: EngineInitialState;
/** Session resume: the full historical messages of the current context (for rendering, including interrupted turns and their markers), for frontend display. */
resumedHistory?: OmniMessage[];
/**
* Set when the session's model doesn't support images (the composition layer decides this via
* ModelEntry.vision): images in `run` input are saved to this directory (the session's
* scratchpad), and the path is appended to the user text instead — the model views the image
* via describe_image, and images never enter the session history directly (some providers
* return a 400 outright on image input).
*/
inputImagesDir?: string;
}
/** Cap on captured title material (chars per side, matching buildTitlePrompt's truncation); stops accumulating once exceeded. */
const TITLE_MATERIAL_LIMIT = 2000;
/**
* Accumulates title material: the body text of complete text messages from the main session
* (no origin) — thinking and tool calls naturally don't count — and stops once the cap is hit.
*/
function appendTitleText(base: string, msg: OmniMessage, role: "user" | "assistant"): string {
if (base.length >= TITLE_MATERIAL_LIMIT) return base;
if (msg.origin && msg.origin.length > 0) return base;
const p = msg.payload as { type?: string; role?: string; text?: string };
if (msg.type !== "model_msg" || p.type !== "text" || p.role !== role || !p.text) return base;
return base ? `${base}\n${p.text}` : p.text;
}
export class Session {
readonly sessionId: string;
/** The session model's provider group (paired with `modelId` to form the model reference). */
readonly provider: string;
/** The session model's upstream model_id (the request id sent to AgentHub). */
readonly modelId: string;
readonly workspaceDir: string;
/** Session resume: the full historical messages of the current context (for rendering); undefined for a non-resumed Session. */
readonly resumedHistory?: OmniMessage[];
private readonly engine: ContextEngine;
private readonly environment: EnvironmentInterface;
private readonly trace?: TraceSink;
private readonly meta: OmniMessage;
private readonly createBareLLM?: () => LLMInterface;
private readonly inputImagesDir?: string;
private metaWritten = false;
/** Title material (used by `generateTitle` as the default): the user input and model body text of the first Task that contains user text. */
private titleUserText = "";
private titleAssistantText = "";
/** Material-frozen flag: becomes true once the first Task containing user text finishes; subsequent runs stop accumulating. */
private titleMaterialFrozen = false;
constructor(config: SessionConfig) {
this.sessionId = config.meta.session_id;
this.provider = config.meta.provider;
this.modelId = config.meta.model_id;
this.workspaceDir = config.meta.workspace;
this.environment = config.environment;
this.trace = config.trace;
this.meta = sessionMeta(config.meta);
this.metaWritten = config.metaAlreadyWritten ?? false;
if (config.resumedHistory) this.resumedHistory = config.resumedHistory;
if (config.createBareLLM) this.createBareLLM = config.createBareLLM;
if (config.inputImagesDir) this.inputImagesDir = config.inputImagesDir;
this.engine = new ContextEngine({
llm: config.llm,
environment: config.environment,
...(config.trace ? { trace: config.trace } : {}),
...(config.maxTurns !== undefined ? { maxTurns: config.maxTurns } : {}),
// Context compaction: new LLM factory + resolved settings + writes session_meta at the start of the new Trace file after splitting.
...(config.createLLM ? { createLLM: config.createLLM } : {}),
...(config.compaction ? { compaction: config.compaction } : {}),
...(config.initialEngineState ? { initialState: config.initialEngineState } : {}),
sessionMeta: this.meta,
});
}
/**
* Runs a Task to completion and streams out OmniMessage. `newMessages` is this call's Prompt
* (only the newly added input); `opts` carries the abort signal `signal` and the per-call
* approval callback `approve` (the engine calls it once per tool_call within a turn).
* On the first run, `session_meta` is written to the Trace first.
*
* A single `run` automatically drives the whole ReAct loop: consuming the LLM stream,
* approving and executing tools one at a time, feeding results back for the next turn,
* until a turn no longer produces a tool_call (Task ends) or it's aborted.
* Docs: /docs/agent-loop § "The loop at a glance".
*/
async *run(newMessages: OmniMessage[], opts?: RunOptions): AsyncGenerator<OmniMessage> {
// Model doesn't support images: input images are saved to disk first (session scratchpad),
// then the path is appended to the text before it reaches the engine/Trace.
if (this.inputImagesDir) {
newMessages = await imagesToScratchpadPaths(newMessages, this.inputImagesDir);
}
await this.ensureMetaWritten();
// Self-captures title material (the title is derived from the first-turn
// conversation text): while material isn't frozen yet, collect this call's user text and
// the produced model text; freezes once the first Task containing user text finishes, so
// the title reflects the start of the conversation.
const capture = !this.titleMaterialFrozen;
if (capture) {
for (const m of newMessages) {
this.titleUserText = appendTitleText(this.titleUserText, m, "user");
}
}
for await (const msg of this.engine.run(newMessages, opts)) {
if (capture) {
this.titleAssistantText = appendTitleText(this.titleAssistantText, msg, "assistant");
}
yield msg;
}
if (capture && this.titleUserText.trim()) this.titleMaterialFrozen = true;
}
/**
* User-initiated request to compact context (e.g. a CLI command): reuses the automatic
* compaction flow but skips the threshold check (reason=manual). Only callable at Task
* boundaries (between runs); streams out paired `compaction` events. The summarize digest
* becomes the prefix of the next `run`'s input (merged with the next user Prompt). A no-op
* if compaction isn't configured.
* Docs: /docs/agent-loop § "Compaction".
*/
async *compact(opts?: { signal?: AbortSignal }): AsyncGenerator<OmniMessage> {
yield* this.engine.compact(opts);
}
/**
* Whether compaction is possible, and why not if not (see ContextEngine.compactability).
* When the result isn't `ok`, `compact()` is a no-op and yields no messages — callers should
* give feedback based on this rather than triggering a silent, fruitless compaction.
*/
compactability(): CompactAvailability {
return this.engine.compactability();
}
/** Writes `session_meta` to the Trace before the first run/compaction; best-effort — failure doesn't interrupt the run. */
private async ensureMetaWritten(): Promise<void> {
if (this.metaWritten) return;
if (this.trace) {
try {
await this.trace.write(this.meta);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
process.stderr.write(`[trace] session_meta write failed: ${message}\n`);
}
}
this.metaWritten = true;
}
/**
* Out-of-band, one-off request that generates a short title from the first-turn conversation
* text: sends one request using the bare LLM for the session's Model (no
* tools, no system prompt, thinking off), **without writing history or Trace**. Material
* defaults to the first Task text self-captured by the Session (user input and model body
* text collected during run; thinking and tool calls don't count), so callers don't need to
* supply it; `material` can override this (e.g. when a host generates a title for a
* sub-session — the material is that sub-session's own conversation). `title` is null if the
* material is empty, the request fails, or the composition layer didn't supply a bare LLM
* factory. Token consumption is returned via `usage` for the host to account for.
* Docs: /docs/agent-loop § "Side channels".
*/
async generateTitle(args?: {
/** Material override; defaults to the Session's self-captured material. */
material?: { userText: string; assistantText: string };
signal?: AbortSignal;
}): Promise<SessionTitleResult> {
if (!this.createBareLLM) return { title: null, usage: null };
const material = args?.material ?? {
userText: this.titleUserText,
assistantText: this.titleAssistantText,
};
return generateTitleWithLLM(this.createBareLLM(), {
...material,
...(args?.signal ? { signal: args.signal } : {}),
});
}
/** Queries a tool's permission level (for the frontend to determine permission mode); returns undefined for unknown tools. */
toolPermission(name: string): ToolPermission | undefined {
return this.environment.toolPermission(name);
}
/** This Session's session_meta message (used e.g. by host tools to forward nested-session metadata to a parent session). */
get metaMessage(): OmniMessage {
return this.meta;
}
/**
* Releases runtime resources held by the Session: kills long-running command sessions
* managed by the Environment. The host calls this when the Session ends (CLI exit, Web
* session close) to avoid leaking background processes into the host process's lifetime.
* Optional, idempotent.
*/
dispose(): void {
this.environment.dispose?.();
}
}
+418
View File
@@ -0,0 +1,418 @@
/**
* Loading and initialization of Agent State (semantics modeled on Hugging Face model loading).
*
* - Initializes when the target Agent directory is empty (no `system_config.yaml`): creates
* `agent_state/`, `tools/`, `memory/`, `skills/`, and the sibling `scratchpad/`, and writes
* the default `system_config.yaml` and `AGENTS.md`.
* - Otherwise loads the existing system config and editable Prompt for the given `agentId`.
*
* The full runtime Prompt is rendered from the system-level Prompt template in
* `system_config.yaml`; placeholders in the template are replaced with `AGENTS.md` and the
* concrete Session runtime environment fields. Built-in tools and MCP Server config
* come from `system_config.yaml`.
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseYaml, stringify as stringifyYaml } from "yaml";
import {
loadLibrarySkills,
parseSkillFrontmatter,
type SkillMetadata,
} from "@prismshadow/penguin-skills";
import type { ToolConfig, ToolDefinitionConfig } from "../interfaces.js";
import {
AGENT_ID_PLACEHOLDER,
AGENTS_MD_PLACEHOLDER,
VAULT_KEYS_PLACEHOLDER,
SKILL_METADATA_PLACEHOLDER,
CWD_PLACEHOLDER,
DATE_PLACEHOLDER,
defaultAgentsMd,
defaultSystemConfig,
OS_VERSION_PLACEHOLDER,
PLATFORM_PLACEHOLDER,
PROJECT_DIR_PLACEHOLDER,
SESSION_ID_PLACEHOLDER,
type SystemConfig,
} from "./default-config.js";
import { builtinProjectAgentPresets, type AgentPreset } from "./builtin-agents.js";
import { provisionExampleBenchmark } from "./example-benchmark.js";
import {
agentsMdPath,
agentStateDir,
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
memoryDir,
resolveRoot,
scratchpadDir,
skillsDir,
systemConfigPath,
toolsDir,
} from "./paths.js";
/** project_id / agent_id / skill_name only allow letters, digits, underscore `_`, and hyphen `-` (prevents path traversal). */
const ID_PATTERN = /^[A-Za-z0-9_-]+$/;
export type IdKind = "project_id" | "agent_id" | "skill_name";
export function isValidId(id: string): boolean {
return ID_PATTERN.test(id);
}
export function assertValidId(kind: IdKind, id: string): void {
if (!ID_PATTERN.test(id)) {
throw new Error(
`Invalid ${kind} ${JSON.stringify(id)}: only letters, digits, "_" and "-" are allowed.`,
);
}
}
/** A loaded Agent State handle. */
export interface AgentState {
root: string;
projectId: string;
agentId: string;
stateDir: string;
systemConfig: SystemConfig;
agentsMd: string;
}
export interface SessionEnvironmentValues {
sessionId: string;
cwd: string;
/** The Agent id this Session belongs to (system Prompt placeholder {{AGENT_ID}}). */
agentId: string;
/** Absolute path to this Project's directory (system Prompt placeholder {{PROJECT_DIR}}; Agent State/scratchpad paths are derived from it). */
projectDir: string;
platform: string;
osVersion: string;
date: string;
}
/**
* Loads or initializes Agent State.
*
* When root/project/agent are omitted, `resolveRoot()` and the default constants are used. If
* `system_config.yaml` doesn't exist, the directory is treated as empty and initialized;
* otherwise the existing content is loaded. `preset` only takes effect on the initialization
* path (name/description/AGENTS.md overrides and extra Skills) and is ignored when loading an
* existing Agent — existing config is never overwritten.
*/
export async function loadOrInitAgentState(opts?: {
agentId?: string;
projectId?: string;
root?: string;
preset?: AgentPreset;
}): Promise<AgentState> {
const root = opts?.root ?? resolveRoot();
const projectId = opts?.projectId ?? DEFAULT_PROJECT_ID;
const agentId = opts?.agentId ?? DEFAULT_AGENT_ID;
// Validate before building paths, to prevent path traversal.
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
const stateDir = agentStateDir(root, projectId, agentId);
const configPath = systemConfigPath(root, projectId, agentId);
const mdPath = agentsMdPath(root, projectId, agentId);
let systemConfig: SystemConfig;
let agentsMd: string;
if (await fileExists(configPath)) {
// Load path: read the existing system_config.yaml and AGENTS.md.
const rawConfig = await fs.readFile(configPath, "utf8");
const parsed = parseYaml(rawConfig) as unknown;
// Defensive check: if the file is empty/corrupted, parseYaml may return null/a non-object,
// or system_prompt may be missing — otherwise "undefined" would get spliced into the system
// Prompt. Throw a clear error when validation fails.
if (
parsed === null ||
typeof parsed !== "object" ||
typeof (parsed as SystemConfig).system_prompt !== "string"
) {
throw new Error(`Agent State 配置非法:${configPath} 为空、损坏或缺少 system_prompt 字段。`);
}
systemConfig = parsed as SystemConfig;
agentsMd = (await fileExists(mdPath)) ? await fs.readFile(mdPath, "utf8") : defaultAgentsMd();
} else {
// Init path: create the directory structure and write default config (preset only takes effect here).
await Promise.all([
fs.mkdir(stateDir, { recursive: true }),
fs.mkdir(toolsDir(root, projectId, agentId), { recursive: true }),
fs.mkdir(memoryDir(root, projectId, agentId), { recursive: true }),
fs.mkdir(skillsDir(root, projectId, agentId), { recursive: true }),
fs.mkdir(scratchpadDir(root, projectId, agentId), { recursive: true }),
]);
const preset = opts?.preset;
systemConfig = {
...defaultSystemConfig(),
...(preset?.name !== undefined ? { name: preset.name } : {}),
...(preset?.description !== undefined ? { description: preset.description } : {}),
};
agentsMd = preset?.agentsMd ?? defaultAgentsMd();
// Only installs the Skills specified by preset (a plain newly created Agent gets none
// pre-installed). A default_agent with no
// preset (e.g. created on first CLI run) still gets every Skill in the library pre-installed
// — the install policy follows Agent identity, not whether creation came from the server or
// was done directly via SDK/CLI.
// Skills have no dedicated tool: metadata is injected via {{SKILL_METADATA}}, and the model
// reads SKILL.md with shell and follows it.
const skills =
opts?.preset === undefined && agentId === DEFAULT_AGENT_ID
? loadLibrarySkills()
: (opts?.preset?.skills ?? []);
await Promise.all([
fs.writeFile(mdPath, agentsMd, "utf8"),
...skills.map((skill) => installSkill(root, projectId, agentId, skill)),
// The example Benchmark is only provisioned alongside default_agent (so the evaluation
// center has data out of the box): idempotently skipped if benchmarks/ already exists,
// and not created for plain Agents.
...(agentId === DEFAULT_AGENT_ID
? [provisionExampleBenchmark(root, projectId, agentId)]
: []),
]);
// system_config.yaml is written last: its existence is the "initialization complete" marker
// (the load/init decision point). If this fails partway (disk full / crash), the next run
// still takes the init path and self-heals, so no half-initialized state with missing Skills is left behind.
await fs.writeFile(configPath, stringifyYaml(systemConfig), "utf8");
}
return { root, projectId, agentId, stateDir, systemConfig, agentsMd };
}
/**
* Initializes a Project's built-in Agent (the only built-in Agent: default_agent).
*
* Calls loadOrInitAgentState for each one: an Agent whose directory already exists (including a
* default_agent created earlier by the CLI) is only loaded, never overwritten (preset only
* takes effect on initialization). Returns the list of built-in Agent ids.
*/
export async function provisionProjectAgents(opts?: {
root?: string;
projectId?: string;
}): Promise<string[]> {
const agentIds: string[] = [];
for (const { agentId, preset } of builtinProjectAgentPresets()) {
await loadOrInitAgentState({
...(opts?.root !== undefined ? { root: opts.root } : {}),
...(opts?.projectId !== undefined ? { projectId: opts.projectId } : {}),
agentId,
preset,
});
agentIds.push(agentId);
}
return agentIds;
}
/**
* The vault key-name list: the replacement value for `{{VAULT_KEYS}}`, one `- KEY` per line;
* returns an empty string when there are no keys.
* **Contains only key names, never values** — values are only injected into the exec_command
* subprocess environment, never the model context. The statement of the vault's purpose is part
* of the default template body (the # Vault section) and is kept even with no vault.
*/
function vaultKeysList(keys: string[]): string {
return keys.map((key) => `- ${key}`).join("\n");
}
/**
* Installs a Skill into the target Agent: writes `skills/<name>/SKILL.md` verbatim (the full
* SKILL.md content including frontmatter, ensuring a trailing newline); if the directory
* already exists, it's overwritten (reinstalling = updating to the latest content). An optional
* icon.svg is written alongside SKILL.md; if this install doesn't
* include an icon, any old icon.svg is removed, preserving "overwrite update" semantics (the
* directory content matches the Skill being installed).
* Docs: /docs/skills § "Installation and storage".
*/
export async function installSkill(
root: string,
projectId: string,
agentId: string,
skill: { name: string; content: string; icon?: string },
): Promise<void> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
assertValidId("skill_name", skill.name);
const dir = path.join(skillsDir(root, projectId, agentId), skill.name);
await fs.mkdir(dir, { recursive: true });
const content = skill.content.endsWith("\n") ? skill.content : `${skill.content}\n`;
const iconPath = path.join(dir, "icon.svg");
await Promise.all([
fs.writeFile(path.join(dir, "SKILL.md"), content, "utf8"),
skill.icon !== undefined
? fs.writeFile(iconPath, skill.icon, "utf8")
: fs.rm(iconPath, { force: true }),
]);
}
/** Uninstalls a Skill: deletes the entire `skills/<name>/` directory; idempotent, no error if it doesn't exist. */
export async function removeSkill(
root: string,
projectId: string,
agentId: string,
name: string,
): Promise<void> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
assertValidId("skill_name", name);
await fs.rm(path.join(skillsDir(root, projectId, agentId), name), {
recursive: true,
force: true,
});
}
/** An installed Skill entry: frontmatter metadata (including an optional short description) + the optional icon.svg content in the directory. */
export interface InstalledSkill extends SkillMetadata {
/** The raw content of `skills/<name>/icon.svg` (a custom icon copied alongside SKILL.md at install time); the field is omitted when missing (the frontend falls back to a default book icon). */
icon?: string;
}
/**
* Lists the metadata of Skills installed on the target Agent: scans `skills/<name>/SKILL.md` and
* parses its frontmatter (optional fields like short_description(_zh) pass through as parsed),
* also reading the optional icon.svg content in the directory. Tolerant: a directory whose
* frontmatter fails to parse or is missing `name` falls back to
* `{ name: <directory name>, description: "", version: 1, updated: "" }`; a directory with no
* SKILL.md doesn't count as a Skill; returns [] if skills/ doesn't exist. Results are sorted by
* name (a stable order for both Prompt injection and API responses).
* Docs: /docs/skills § "Installation and storage".
*/
export async function listInstalledSkills(
root: string,
projectId: string,
agentId: string,
): Promise<InstalledSkill[]> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
const dir = skillsDir(root, projectId, agentId);
let entries;
try {
entries = await fs.readdir(dir, { withFileTypes: true });
} catch {
return [];
}
const skills: InstalledSkill[] = [];
for (const entry of entries) {
if (!entry.isDirectory()) continue;
let raw: string;
try {
raw = await fs.readFile(path.join(dir, entry.name, "SKILL.md"), "utf8");
} catch {
continue;
}
let icon: string | undefined;
try {
icon = await fs.readFile(path.join(dir, entry.name, "icon.svg"), "utf8");
} catch {
// icon.svg is optional: missing means no custom icon.
}
// The directory name is the Skill's identity (install / uninstall / Prompt read guidance all
// address by directory name): frontmatter only supplies display fields like description; when
// its `name` doesn't match the directory name (a hand-written or network-sourced Skill), the
// directory name always wins — otherwise the model would read a nonexistent path using the
// injected name, and the API couldn't uninstall it either.
const parsed = parseSkillFrontmatter(raw);
skills.push({
...(parsed ?? { description: "", version: 1, updated: "" }),
name: entry.name,
...(icon !== undefined ? { icon } : {}),
});
}
return skills.sort((a, b) => a.name.localeCompare(b.name));
}
/**
* Skill metadata section: the replacement value for `{{SKILL_METADATA}}`, one line per Skill in
* the form `- \`name\` — description` (just the name when description is empty); an empty array
* returns an empty string. The full body is read by the model on demand via shell.
*/
export function skillMetadataSection(skills: SkillMetadata[]): string {
return skills
.map((s) => (s.description ? `- \`${s.name}\` — ${s.description}` : `- \`${s.name}\``))
.join("\n");
}
/**
* Renders the complete runtime system Prompt: substitutes `AGENTS.md`, vault key names, Skill
* metadata, and the concrete Session runtime environment placeholders into the system Prompt
* template. The assembly layer only does placeholder substitution and adds no extra text —
* wrapper text such as `<developer_instructions>` and the # Vault / # Skills statements are
* written directly into the system Prompt template itself (the Prompt is fully
* transparent and editable via `system_config.yaml`). Other files in Agent State / Workspace are
* never auto-injected.
*
* `{{VAULT_KEYS}}` is replaced with the vault key-name list (an empty string if empty/not
* provided): this lets the model know which APIs requiring a key it can call; values are never
* injected. `{{SKILL_METADATA}}` is replaced with the installed Skills' metadata lines (an empty
* string if empty/not provided). A custom template that removes a placeholder gets no
* corresponding content injected.
* Docs: /docs/configuration § "System prompt placeholders".
*/
export function assembleSystemPrompt(
state: AgentState,
sessionEnvironment?: SessionEnvironmentValues,
vaultKeys?: string[],
skillMetadata?: SkillMetadata[],
): string {
return state.systemConfig.system_prompt
.split(AGENTS_MD_PLACEHOLDER)
.join(state.agentsMd.trim())
.split(VAULT_KEYS_PLACEHOLDER)
.join(vaultKeysList(vaultKeys ?? []))
.split(SKILL_METADATA_PLACEHOLDER)
.join(skillMetadataSection(skillMetadata ?? []))
.split(AGENT_ID_PLACEHOLDER)
.join(sessionEnvironment?.agentId ?? state.agentId)
.split(PROJECT_DIR_PLACEHOLDER)
.join(sessionEnvironment?.projectDir ?? "")
.split(SESSION_ID_PLACEHOLDER)
.join(sessionEnvironment?.sessionId ?? "")
.split(CWD_PLACEHOLDER)
.join(sessionEnvironment?.cwd ?? "")
.split(PLATFORM_PLACEHOLDER)
.join(sessionEnvironment?.platform ?? "")
.split(OS_VERSION_PLACEHOLDER)
.join(sessionEnvironment?.osVersion ?? "")
.split(DATE_PLACEHOLDER)
.join(sessionEnvironment?.date ?? "")
.trim();
}
/**
* Builds the `ToolConfig` needed by Environment from Agent State.
*
* Both builtin tools and MCP Server config are taken from `system_config.yaml`; falls back to the
* default config when builtin tools are missing.
*/
/**
* Filters builtin tool entries by the session model's type: entries with `forModel: "vision"` are
* only used for models that support images (vision models), `forModel: "text-only"` is only for
* text-only models (e.g. choosing between read_image / describe_image); unlabeled entries are
* available to all models.
* Docs: /docs/tools § "Image tools".
*/
export function selectBuiltinToolsForModel(
tools: ToolDefinitionConfig[],
modelVision: boolean,
): ToolDefinitionConfig[] {
const kind = modelVision ? "vision" : "text-only";
return tools.filter((t) => t.forModel === undefined || t.forModel === kind);
}
export function buildToolConfig(state: AgentState): ToolConfig {
const systemTools = state.systemConfig.tools;
const builtin = systemTools?.builtin ?? defaultSystemConfig().tools?.builtin ?? [];
return {
customTools: builtin,
mcpServers: systemTools?.mcpServers ?? [],
};
}
async function fileExists(filePath: string): Promise<boolean> {
try {
await fs.access(filePath);
return true;
} catch {
return false;
}
}
+146
View File
@@ -0,0 +1,146 @@
/**
* Agent-level environment-variable vault (`<project>/agents/<agent_id>/agent_state/.vault.toml`).
*
* Key-value pairs such as third-party API keys, configured per Agent: injected into that Agent
* session's `exec_command` / `input_command` child-process environment, with key names disclosed
* to the model via the system Prompt while values never enter the model context. Carries the same
* trade-offs as a credential: stored in plaintext on disk, masked at the API layer. The file is
* created/removed together with the Agent directory; its absence is treated as an empty table;
* once emptied, the file is removed to avoid leaving a stray empty .vault.toml.
* Docs: /docs/configuration § "Vault".
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseToml, stringify as stringifyToml } from "smol-toml";
import { agentVaultPath } from "./paths.js";
import { assertValidId } from "./agent-state.js";
/** Vault key-name constraint: matches shell environment variable names (starts with a letter or underscore, followed by letters/digits/underscores only). */
const VAULT_KEY_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/;
/**
* Vault value length cap: since values are injected into the child-process environment, Linux
* caps a single env entry at roughly 128KB, and an oversized value would make every
* exec_command spawn for that Agent fail (E2BIG) — so it's rejected on the write side (core and
* the API layer share this same cap).
*/
export const VAULT_VALUE_MAX_LENGTH = 8192;
/** Whether a vault key name is valid (shell environment variable name rules). */
export function isValidVaultKey(key: string): boolean {
return VAULT_KEY_PATTERN.test(key);
}
/** Validates a vault key name, throwing if invalid (core and the API layer share this same rule). */
export function assertValidVaultKey(key: string): void {
if (!isValidVaultKey(key)) {
throw new Error(
`Invalid vault key ${JSON.stringify(key)}: only letters, digits and "_" are allowed, and it must not start with a digit.`,
);
}
}
/** Validates a vault value's length (see `VAULT_VALUE_MAX_LENGTH`), throwing if it exceeds the cap. */
export function assertValidVaultValue(key: string, value: string): void {
if (value.length > VAULT_VALUE_MAX_LENGTH) {
throw new Error(
`Vault value for ${key} is too long: ${value.length} > ${VAULT_VALUE_MAX_LENGTH} characters.`,
);
}
}
/**
* Reads the Agent vault: returns an empty table if the file doesn't exist.
* A hand-edited file is filtered by the same rule as the write side: only string values are
* accepted (numbers/dates etc. are ignored), and key names must follow shell variable name rules
* (invalid keys are always ignored — otherwise they'd get injected into the Prompt/child-process
* environment, and an invalid key surfaced by a GET view would make a full-table PUT 400, leaving
* the vault page unable to add or remove any further entries).
*/
export async function loadAgentVault(
root: string,
projectId: string,
agentId: string,
): Promise<Record<string, string>> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
let raw: string;
try {
raw = await fs.readFile(agentVaultPath(root, projectId, agentId), "utf8");
} catch {
return {};
}
const parsed: unknown = parseToml(raw) ?? {};
const vault: Record<string, string> = {};
if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) {
for (const [k, v] of Object.entries(parsed)) {
if (typeof v === "string" && isValidVaultKey(k)) vault[k] = v;
}
}
return vault;
}
/**
* Writes the full table to the Agent vault: validates all key names first; an empty table
* deletes the file (idempotent if it doesn't exist).
* The directory is created automatically if it doesn't exist (the vault can be configured even
* before the Agent is initialized).
*/
export async function saveAgentVault(
root: string,
projectId: string,
agentId: string,
vault: Record<string, string>,
): Promise<void> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
for (const key of Object.keys(vault)) assertValidVaultKey(key);
const file = agentVaultPath(root, projectId, agentId);
if (Object.keys(vault).length === 0) {
await fs.rm(file, { force: true });
return;
}
await fs.mkdir(path.dirname(file), { recursive: true });
// The secret file is written to disk with mode 0600 (a hidden file blocks `ls`, not reads; mode
// only takes effect on creation, so chmod is applied to converge an existing file too).
await fs.writeFile(file, `${stringifyToml(vault)}\n`, { encoding: "utf8", mode: 0o600 });
await fs.chmod(file, 0o600);
}
/**
* Writes or updates one vault entry (added if it doesn't exist, overwritten if it does).
* The key name must follow shell environment variable name rules (see `isValidVaultKey`), and the
* value length is constrained by `VAULT_VALUE_MAX_LENGTH`; throws if invalid. Returns the updated
* vault.
*/
export async function setVaultEntry(
root: string,
projectId: string,
agentId: string,
key: string,
value: string,
): Promise<Record<string, string>> {
assertValidVaultKey(key);
assertValidVaultValue(key, value);
const vault = await loadAgentVault(root, projectId, agentId);
vault[key] = value;
await saveAgentVault(root, projectId, agentId, vault);
return vault;
}
/**
* Removes one vault entry; idempotent if the key doesn't exist (no write happens). Once emptied,
* the whole .vault.toml is removed. Returns the updated vault.
*/
export async function removeVaultEntry(
root: string,
projectId: string,
agentId: string,
key: string,
): Promise<Record<string, string>> {
const vault = await loadAgentVault(root, projectId, agentId);
if (!(key in vault)) return vault;
delete vault[key];
await saveAgentVault(root, projectId, agentId, vault);
return vault;
}
+48
View File
@@ -0,0 +1,48 @@
/**
* Preset content for builtin Agents; Skill documentation lives in @prismshadow/penguin-skills
* (the library files are read live when building the preset).
*
* - Every Project comes with a single builtin Agent: `default_agent` (the General Agent, the
* default conversational Agent), which has every Skill in the library installed at
* initialization. Dedicated capabilities (creating an Agent, optimizing an Agent, etc.)
* are carried by Skills rather than dedicated builtin Agents.
* - The preset carries no AGENTS.md: the default AGENTS.md is empty, with delegation and task
* conventions living in the default template's Suggested Workflows section.
* - Skill metadata is auto-injected into the system Prompt via the `{{SKILL_METADATA}}`
* placeholder; it's not registered in AGENTS.md.
*/
import { loadLibrarySkills, type LibrarySkill } from "@prismshadow/penguin-skills";
import { DEFAULT_AGENT_ID } from "./paths.js";
/** The set of Project builtin Agent ids (supplied along with the Project, cannot be deleted from Web). */
export const BUILTIN_AGENT_IDS: readonly string[] = [DEFAULT_AGENT_ID];
/** Agent initialization preset (only takes effect at initialization; ignored when loading an existing Agent). */
export interface AgentPreset {
/** Display name written to system_config.yaml. */
name?: string;
/** Description written to system_config.yaml. */
description?: string;
/** Overrides the default AGENTS.md content. */
agentsMd?: string;
/** Skills installed at initialization (installs none by default). */
skills?: LibrarySkill[];
}
/**
* The preset list for a Project's builtin Agents (each initialized in turn when the Project is
* created; an existing Agent is never overwritten). The only builtin Agent is default_agent:
* installs every Skill in the library, with no preset AGENTS.md.
*/
export function builtinProjectAgentPresets(): Array<{ agentId: string; preset: AgentPreset }> {
return [
{
agentId: DEFAULT_AGENT_ID,
preset: {
name: "General Agent",
description: "General-purpose agent that completes the user's requests with its tools.",
skills: loadLibrarySkills(),
},
},
];
}
+393
View File
@@ -0,0 +1,393 @@
/**
* Default system configuration for Agent State (written to `system_config.yaml`) and the
* default `AGENTS.md` (empty).
*
* Runtime Prompt and tool configuration should come from editable files;
* code only supplies the initial defaults. `system_config.yaml` holds the relatively stable
* system-level Prompt, built-in tools, and MCP Server configuration; `AGENTS.md` is injected
* via a system Prompt placeholder.
*
* The system Prompt is sectioned and trimmed as needed (Role/Personality/Success
* criteria/Constraints/Stop rules/File system/Suggested workflows); it does not describe
* specific tools (that comes from the tool schema). AGENTS.md, Vault/Skills, and Environment
* injection go at the end.
*
* Placeholders (`{{...}}`) appear only in the trailing injection zones (AGENTS.md / Vault /
* Skills / Environment); elsewhere the body uses angle-bracket notation such as
* \`<project_dir>\`, \`<agent_id>\`, \`<session_id>\` — these are **not substituted**; the model
* fills in the actual values from the Environment section itself.
*/
import type { MCPServerConfig, ThinkingLevelName, ToolDefinitionConfig } from "../interfaces.js";
import type { CompactionMode } from "../omnimessage/types.js";
/** Docs: /docs/configuration § "System prompt placeholders". */
export const AGENTS_MD_PLACEHOLDER = "{{AGENTS_MD}}";
export const VAULT_KEYS_PLACEHOLDER = "{{VAULT_KEYS}}";
export const SKILL_METADATA_PLACEHOLDER = "{{SKILL_METADATA}}";
export const SESSION_ID_PLACEHOLDER = "{{SESSION_ID}}";
export const CWD_PLACEHOLDER = "{{CWD}}";
export const AGENT_ID_PLACEHOLDER = "{{AGENT_ID}}";
export const PROJECT_DIR_PLACEHOLDER = "{{PROJECT_DIR}}";
export const PLATFORM_PLACEHOLDER = "{{PLATFORM}}";
export const OS_VERSION_PLACEHOLDER = "{{OS_VERSION}}";
export const DATE_PLACEHOLDER = "{{DATE}}";
/**
* Context compaction config (the `compaction` section of `system_config.yaml`).
* Docs: /docs/configuration § "Agent config".
*/
export interface CompactionConfig {
/** Context Token threshold (taken from the most recent token_usage's request.total); defaults to 128000, <=0 disables. */
max_context_length?: number;
/** Session cumulative turn threshold (counted in LLM Requests, across Tasks); defaults to -1, <=0 means no limit. */
max_session_turns?: number;
/** Compaction mode; defaults to summarize. */
mode?: CompactionMode;
/** Prompt template for summarize compaction; defaults to the built-in value (editable config, not hardcoded). */
prompt?: string;
}
/**
* System-level config for Agent State, serialized as `system_config.yaml`.
* Docs: /docs/configuration § "Agent config".
*/
export interface SystemConfig {
/** Agent display name (display name is separate from id; falls back to id when unset). */
name?: string;
/** Agent description. */
description?: string;
/** Agent State version number: a natural number, 1 on creation, incremented on successful optimization; a missing field is treated as 1. */
version?: number;
/** System-level Prompt (relatively stable; should not be modified frequently). */
system_prompt: string;
/** Max LLM turns per Task (a runtime parameter that belongs to Agent config, not specified when creating a Session). */
max_turns?: number;
model?: {
max_tokens?: number;
thinking_level?: ThinkingLevelName;
timeoutMs?: number;
};
/** Context compaction (enabled by default, max_context_length 128k, mode summarize). */
compaction?: CompactionConfig;
tools?: {
/** Built-in system tool configuration. */
builtin?: ToolDefinitionConfig[];
/** MCP Server configuration. */
mcpServers?: MCPServerConfig[];
};
}
const DEFAULT_SYSTEM_PROMPT = `# Role
You are PenguinHarness, an agent that completes the user's requests on their machine with the tools available to you.
# Personality
Communicate with the user precisely and concisely, yet with warmth. Do not repeatedly explain your tools or restate their results.
# Success criteria
- Before delivering the result, check that every problem in the request has been solved.
- Verify your work through every available means; never claim a result you did not observe.
# Constraints
- Make the smallest change that satisfies the request; do not modify unrelated files.
- Destructive operations are forbidden.
- Never kill a process you did not start yourself (e.g. to free a busy port) unless the user explicitly asks you to.
- If a tool call fails, read the error, adjust, and retry; never repeat the same failing input.
# Stop rules
- Stop and give the final answer once the success criteria are met.
- If the request is ambiguous, stop and ask the user for clarification instead of guessing their intent.
- If you hit an error you cannot resolve, stop and report the blocker to the user.
# Tool use
- Prefer solving problems with your tools: inspect the real files and environment and run real commands instead of answering from memory or guessing.
- When you need information from the internet, browse it with your shell tool — \`curl\` for pages and APIs, or Playwright (if installed) for dynamic sites.
# System markers
Some user-side messages are system-synthesized records, not user text to answer directly:
- \`<turn_aborted>\`: the previous round was interrupted. Inside are the original request, your partial thinking/text, and the tool calls already issued with their results. Continue from where it left off; do not re-run tools whose results are already included.
- \`<turn_retried>\`: the previous attempt of this round failed on a transport error (timeout or malformed response) — the user did NOT interrupt — and this request is the automatic retry. Inside are your partial thinking/text and the tool calls already executed with their results. Continue from them; do not re-run tools whose results are already included.
- \`<context_summary>\`: earlier conversation was compacted. This summary replaced the raw transcript and is its only record; treat it as established context and continue the task from it.
# File system
- Angle-bracket markers such as \`<project_dir>\`, \`<agent_id>\` and \`<session_id>\` are not literal paths — substitute the matching values from the Environment section.
- You run inside the user's working folder (\`CWD\` in Environment).
- The project directory is \`<project_dir>\`; every agent of this project lives under \`<project_dir>/agents/\`, so another agent's assets are at \`<project_dir>/agents/<its_agent_id>/agent_state/\`.
- Your own Agent State is \`<project_dir>/agents/<agent_id>/agent_state/\` — it holds your assets such as \`skills/\`, and its \`AGENTS.md\` is already included in your context. Reach these paths directly.
- For temporary and scratch files, create a subdirectory named after the current Session ID under your scratchpad: \`<project_dir>/agents/<agent_id>/scratchpad/<session_id>/\`. Build intermediates there, but always place final deliverables in the workspace (under \`CWD\`) — files left in the scratchpad are not part of your output.
- When you create or update a file in the workspace, mention its workspace-relative path in backticks (e.g. \`src/app.py\`) in your reply, so the user can open it from the message.
- Never read, copy, print or otherwise access \`<project_dir>/.project_config.toml\` or any agent's \`agent_state/.vault.toml\` — they hold the user's API keys and other secrets, which are none of your business. Configuration is CLI-only: change models or credentials with \`penguin config ...\` commands. If a task seems to require these files, say so and ask the user instead.
# Suggested workflows
These are recommendations, not requirements; adapt them as the task demands.
- For a long-horizon task, first write a plan in Markdown to \`<project_dir>/agents/<agent_id>/scratchpad/<session_id>/PLAN.md\`, containing a task overview and an itemized step-by-step plan; update it after each completed step to keep execution consistent.
- Delegate self-contained subtasks to other agents with the \`run_subagent\` tool; dispatch independent subtasks in parallel. Start every delegation prompt with your own agent id (e.g. "Caller agent: <agent_id>") and name the skill the subagent should use when the task matches one. Subagents share your Workspace — exchange data through files. If \`run_subagent\` is not in your tool list, you are the subagent: do the work yourself.
- To visit web pages, prefer Playwright when installed; otherwise \`curl\`. When building a web app or frontend, prefer React.
<developer_instructions>
Custom instructions from the developer-editable AGENTS.md.
{{AGENTS_MD}}
</developer_instructions>
# Vault
The vault holds this agent's per-agent secrets (agent_state/.vault.toml). Each entry is injected into your shell subprocesses as an environment variable — values never appear in your context. Use the variable names below in commands when a task needs them.
{{VAULT_KEYS}}
# Skills
Skills are reusable instruction packages stored under <project_dir>/agents/<agent_id>/agent_state/skills/<skill_name>/SKILL.md. There is no skill tool: when a task matches an installed skill below, or the user asks to use one (a message may start with a <use_skills> block listing skill names), first read that skill's SKILL.md in full with a shell command, then follow it. If a request only names a skill without a concrete task, ask the user what they need before starting.
{{SKILL_METADATA}}
# Environment
- Platform: {{PLATFORM}}
- OS Version: {{OS_VERSION}}
- Date: {{DATE}}
- CWD: {{CWD}}
- Agent ID: {{AGENT_ID}}
- Project Dir: {{PROJECT_DIR}}
- Session ID: {{SESSION_ID}}`;
/**
* Built-in default compaction Prompt (summarize mode): tells the model that after
* compaction the raw transcript is no longer visible and the
* summary is the only record, so it must include everything needed to continue the task,
* and no tools may be called while writing the summary.
*/
export const DEFAULT_COMPACTION_PROMPT =
"You have a partial transcript of the task above. Write a summary of it wrapped in " +
"`<summary></summary>` tags. This summary will replace the transcript: in the next " +
"context window the raw transcript above will no longer be visible and this summary " +
"will be its only record, so include everything needed to continue the task — the " +
"original request, current state, next steps, and any learnings. Do not call any " +
"tools while writing the summary; respond with text only.";
/**
* Default built-in system tools: bash execution and subagent spawning.
* Docs: /docs/tools § "Built-in tools".
*/
function defaultBuiltinTools(): ToolDefinitionConfig[] {
return [
{
name: "exec_command",
description:
"Run a shell command in the workspace to read, write, edit files and run programs. " +
"Run long-lived commands (servers, watchers, builds) in the foreground: past yield_time_ms " +
"they keep running in the background with a process_id. Do not background them with `&` — " +
"the whole process group is cleaned up when the foreground command exits.",
parameters: {
type: "object",
properties: {
cmd: {
type: "string",
description: "Shell command to execute.",
},
workdir: {
type: "string",
description:
"Working directory for the command; defaults to the cwd. Optionally a path relative to the cwd, or an absolute path.",
},
yield_time_ms: {
type: "number",
description:
"How long to wait for the command before yielding. If it is still running when this elapses, the tool returns the output so far plus a process_id, and the command keeps running in the background (drive it with input_command). Defaults to 60000; minimum 250, capped below the tool timeout.",
},
},
required: ["cmd"],
},
permission: "rw",
timeoutMs: 120000,
maxOutputLength: 16000,
},
{
name: "input_command",
description:
"Interact with a running command session started by exec_command: write to its stdin, send Ctrl-C, or poll for new output. Identify the session with its process_id.",
parameters: {
type: "object",
properties: {
process_id: {
type: "string",
description: "The process_id returned by exec_command for the running command session.",
},
chars: {
type: "string",
description:
'Characters to write to the command\'s stdin. Send "\\u0003" alone to deliver Ctrl-C (SIGINT); mixing it with other characters is an error. Empty (the default) writes nothing and only polls for new output and exit status.',
},
yield_time_ms: {
type: "number",
description:
"How long to wait for new output or exit before returning. Non-empty writes default to 250; empty polls default to 5000. Minimum 250, capped below the tool timeout.",
},
},
required: ["process_id"],
},
permission: "rw",
// An empty poll can wait out a build/test run (the yield ceiling is derived from timeoutMs, clamped inside the tool).
timeoutMs: 130000,
maxOutputLength: 16000,
},
{
name: "run_subagent",
description:
"Delegate a self-contained subtask to a subagent that runs autonomously in the same workspace and returns its final answer. Use it for focused sub-tasks you can fully specify in one prompt. Optionally choose a specific agent via `agent_id` and a model via `model_id`. " +
'Begin the prompt by identifying yourself with your own agent id (from the Environment section), e.g. "Caller agent: default_agent" — the subagent cannot otherwise tell who invoked it.',
parameters: {
type: "object",
properties: {
prompt: {
type: "string",
description:
"The complete task for the subagent: include all context it needs and the exact final output you expect back.",
},
agent_id: {
type: "string",
description:
"Which agent to run as the subagent; defaults to the current agent when omitted.",
},
model_id: {
type: "string",
description:
"Which model the subagent should use; defaults to the Project default model when omitted.",
},
yield_time_ms: {
type: "number",
description:
"How long to wait for the subagent before yielding. If it is still working when this elapses, the tool returns the output so far plus a subagent_id, and the subagent keeps running in the background (drive it with input_subagent). Defaults to 300000; minimum 250, capped below the tool timeout.",
},
},
required: ["prompt"],
},
permission: "rw",
// Subagent tasks typically run far longer than a single command, so the timeout ceiling is raised accordingly.
timeoutMs: 600000,
maxOutputLength: 16000,
},
{
name: "input_subagent",
description:
"Interact with a background subagent started by run_subagent: poll for new output, or send a follow-up prompt once it is idle to continue the same subagent session. Identify the session with its subagent_id. Pending tool approvals of the subagent are surfaced while this tool is waiting.",
parameters: {
type: "object",
properties: {
subagent_id: {
type: "string",
description: "The subagent_id returned by run_subagent for the background subagent.",
},
prompt: {
type: "string",
description:
"A follow-up task for the subagent, delivered as a new user message on the same session. Only accepted when the subagent is idle (its previous run finished). Empty (the default) sends nothing and only polls for new output and status.",
},
yield_time_ms: {
type: "number",
description:
"How long to wait for new output or completion before returning. Follow-up prompts default to 300000; empty polls default to 10000. Minimum 250, capped below the tool timeout.",
},
},
required: ["subagent_id"],
},
permission: "rw",
// Same generous timeout tier as run_subagent: an empty poll can wait a long time for the subagent to wrap up.
timeoutMs: 600000,
maxOutputLength: 16000,
},
// The image-reading tools are mutually exclusive based on the session model's type
// (marked via each entry's forModel, filtered at assembly time): read_image is designed
// for vision models (the image is fed back as image content); describe_image is designed
// for text-only models (the image plus the prompt are sent to the Project's configured
// vision model, vision_model, whose text answer becomes the tool output).
{
name: "read_image",
forModel: "vision",
description:
"Read an image and return it as image content for you to view. Accepts an http(s) URL " +
"or a local file path (relative paths resolve against the workspace). " +
"Supports png/jpeg/gif/webp up to 5MB.",
parameters: {
type: "object",
properties: {
source: {
type: "string",
description:
"Image to read: an http(s) URL, or a local file path (absolute, or relative to the workspace).",
},
},
required: ["source"],
},
permission: "r",
timeoutMs: 60000,
maxOutputLength: 16000,
},
{
name: "describe_image",
forModel: "text-only",
description:
"Describe an image and return a TEXT description of it. The current model does not accept " +
"images directly, so the image is analyzed by the project's configured vision model and " +
"you get its text answer back. Use `prompt` to ask exactly what you need to know about " +
"the image (e.g. transcribe text, describe a chart, locate a UI element). Accepts an " +
"http(s) URL or a local file path (relative paths resolve against the workspace). " +
"Supports png/jpeg/gif/webp up to 5MB.",
parameters: {
type: "object",
properties: {
source: {
type: "string",
description:
"Image to read: an http(s) URL, or a local file path (absolute, or relative to the workspace).",
},
prompt: {
type: "string",
description:
"What to ask about the image; the vision model answers this. Defaults to a detailed description.",
},
},
required: ["source"],
},
permission: "r",
// Includes one vision-model request, so the timeout is slightly wider than plain image reading.
timeoutMs: 90000,
maxOutputLength: 16000,
},
];
}
/** Agent State version number: an invalid or missing field is always treated as 1. */
export function agentStateVersion(config: Pick<SystemConfig, "version">): number {
const v = config.version;
return typeof v === "number" && Number.isInteger(v) && v >= 1 ? v : 1;
}
/** Returns the default system configuration for Agent State. */
export function defaultSystemConfig(): SystemConfig {
return {
version: 1,
system_prompt: DEFAULT_SYSTEM_PROMPT,
max_turns: 100,
model: {
max_tokens: 32000,
thinking_level: "medium",
timeoutMs: 120000,
},
compaction: {
max_context_length: 128000,
max_session_turns: -1,
mode: "summarize",
prompt: DEFAULT_COMPACTION_PROMPT,
},
tools: {
builtin: defaultBuiltinTools(),
mcpServers: [],
},
};
}
/**
* Returns the default editable `AGENTS.md` content: an empty string — no guidance is
* preprovisioned by default; Subagent delegation conventions and general task practices
* live in the default template's Suggested workflows section as a soft convention.
* Kept so initialization can still write an empty AGENTS.md file.
*/
export function defaultAgentsMd(): string {
return "";
}
@@ -0,0 +1,345 @@
/**
* Provisioning of the example Benchmark.
*
* When default_agent is initialized, it preprovisions `benchmarks/example-benchmark/`: two
* sample cases (each with statement/ and rubric/ indexed by a README.md),
* benchmark_config.toml (runs = 2), and a scoreboard.yaml with three sample evaluations —
* so the evaluation center has data out of the box. Its description states plainly that this
* is a built-in example and the whole directory can be deleted or replaced. Only
* default_agent gets this; ordinary Agents do not.
*
* Scoring numbers are self-consistent: each case's score / cost / duration_ms is the
* **average** computed from its runs array, and each evaluation's totals are the sum over
* its cases (written this way so it already satisfies the scoreboard v2 convention, and
* tests can verify it).
*/
import fs from "node:fs/promises";
import path from "node:path";
import { stringify as stringifyToml } from "smol-toml";
import { stringify as stringifyYaml } from "yaml";
import { benchmarksDir } from "./paths.js";
/** Directory name of the example Benchmark (the directory name is also its identifier). */
export const EXAMPLE_BENCHMARK_ID = "example-benchmark";
/** Contents of benchmark_config.toml (no model reference here — the model is recorded on each evaluation instead). */
const EXAMPLE_BENCHMARK_CONFIG = {
title: "Example Benchmark",
description:
"A built-in example benchmark so the evaluation charts have data out of the box. " +
"Replace it with your own.",
runs: 2,
};
/** Two sample cases: statement and scoring rubric (in English, 3-5 lines each). */
const EXAMPLE_CASES: Array<{ id: string; statement: string; rubric: string }> = [
{
id: "CASE-001-file-summary",
statement: `# Task: Summarize a project file
Read the provided \`notes.txt\` in your workspace and write \`summary.md\` containing:
1. A one-paragraph overview of at most 3 sentences.
2. A bullet list of the three most important facts.
Keep the whole summary under 150 words.
`,
rubric: `# Scoring rubric (max 5 points)
- 2 pts: \`summary.md\` exists and stays under 150 words.
- 2 pts: The three bullet facts are accurate and taken from \`notes.txt\`.
- 1 pt: The overview paragraph is coherent and at most 3 sentences.
Award partial credit per item; the case score is the sum.
`,
},
{
id: "CASE-002-data-cleanup",
statement: `# Task: Clean up a CSV dataset
The workspace contains \`users.csv\` with duplicate rows and inconsistent casing in the email column.
Produce \`users_clean.csv\` where:
1. Emails are lowercased and rows with an empty email are removed.
2. Exact duplicate rows are dropped, keeping the first occurrence.
Do not change the column order.
`,
rubric: `# Scoring rubric (max 5 points)
- 2 pts: \`users_clean.csv\` exists and keeps the original column order.
- 2 pts: Emails are lowercased, empty-email rows removed, duplicates dropped (first kept).
- 1 pt: No unrelated rows or columns were modified.
Award partial credit per item; the case score is the sum.
`,
},
];
/** Raw result of a single run (a runs element in scoreboard v2). */
interface ExampleRun {
score: number;
cost: number;
duration_ms: number;
session_id: string;
}
/**
* Raw runs for the three sample evaluations (case-level and evaluation-level metrics are
* computed from these, keeping the numbers self-consistent). Each carries the model actually
* used for that round (paired, since the evaluation center's chart splits series by model);
* the examples all use deepseek-v4-pro (a single model, single series).
*/
const EXAMPLE_EVALUATIONS: Array<{
time: string;
version: number;
provider: string;
model_id: string;
summary_title: string;
summary: string;
cases: Array<{ case: string; runs: ExampleRun[] }>;
}> = [
{
time: "2026-07-14T09:30:00Z",
version: 1,
provider: "deepseek",
model_id: "deepseek-v4-pro",
summary_title: "Baseline before any optimization",
summary:
"Example data (not a real evaluation): baseline scores of the built-in sample " +
"benchmark before any optimization. Hypothesis for the next round: the agent skips " +
"a final self-check, losing points on completeness.",
cases: [
{
case: "CASE-001-file-summary",
runs: [
{
score: 2.5,
cost: 0.012,
duration_ms: 42000,
session_id: "session-2026-07-14-09-05-11-1a2b3c01",
},
{
score: 3.5,
cost: 0.014,
duration_ms: 48000,
session_id: "session-2026-07-14-09-13-27-1a2b3c02",
},
],
},
{
case: "CASE-002-data-cleanup",
runs: [
{
score: 3.0,
cost: 0.018,
duration_ms: 66000,
session_id: "session-2026-07-14-09-21-45-1a2b3c03",
},
{
score: 3.0,
cost: 0.022,
duration_ms: 74000,
session_id: "session-2026-07-14-09-28-52-1a2b3c04",
},
],
},
],
},
{
time: "2026-07-15T09:30:00Z",
version: 2,
provider: "deepseek",
model_id: "deepseek-v4-pro",
summary_title: "Added an explicit planning step",
summary:
"Example data (not a real evaluation): after adding an explicit planning step to the " +
"system prompt (hypothesis: written plans reduce missed requirements), both cases " +
"improved. Next: tighten output formatting.",
cases: [
{
case: "CASE-001-file-summary",
runs: [
{
score: 3.5,
cost: 0.011,
duration_ms: 39000,
session_id: "session-2026-07-15-09-04-33-2b3c4d01",
},
{
score: 4.0,
cost: 0.013,
duration_ms: 45000,
session_id: "session-2026-07-15-09-12-08-2b3c4d02",
},
],
},
{
case: "CASE-002-data-cleanup",
runs: [
{
score: 3.5,
cost: 0.016,
duration_ms: 60000,
session_id: "session-2026-07-15-09-19-40-2b3c4d03",
},
{
score: 4.0,
cost: 0.02,
duration_ms: 68000,
session_id: "session-2026-07-15-09-26-59-2b3c4d04",
},
],
},
],
},
{
time: "2026-07-16T09:30:00Z",
version: 3,
provider: "deepseek",
model_id: "deepseek-v4-pro",
summary_title: "Verify deliverables before finishing",
summary:
"Example data (not a real evaluation): after instructing the agent to verify its " +
"deliverables against the statement before finishing (hypothesis: a final check " +
"catches formatting slips), scores improved again. Replace this benchmark with your " +
"own to track real progress.",
cases: [
{
case: "CASE-001-file-summary",
runs: [
{
score: 4.0,
cost: 0.01,
duration_ms: 36000,
session_id: "session-2026-07-16-09-03-21-3c4d5e01",
},
{
score: 4.5,
cost: 0.012,
duration_ms: 40000,
session_id: "session-2026-07-16-09-10-46-3c4d5e02",
},
],
},
{
case: "CASE-002-data-cleanup",
runs: [
{
score: 4.5,
cost: 0.015,
duration_ms: 55000,
session_id: "session-2026-07-16-09-18-02-3c4d5e03",
},
{
score: 4.0,
cost: 0.017,
duration_ms: 61000,
session_id: "session-2026-07-16-09-25-30-3c4d5e04",
},
],
},
],
},
];
/** Round floats to 1e-6 (so binary error from averaging/summing isn't persisted to disk). */
function round(v: number): number {
return Math.round(v * 1e6) / 1e6;
}
function average(values: number[]): number {
return round(values.reduce((a, b) => a + b, 0) / values.length);
}
function sum(values: number[]): number {
return round(values.reduce((a, b) => a + b, 0));
}
/**
* Builds the scoreboard object from raw runs data: each case's three metrics are the
* average of its runs, and each evaluation's metrics are the sum of its cases' averages
* (following the scoreboard v2 convention). Exported so tests can verify the numbers
* are self-consistent.
*/
export function buildExampleScoreboard(): {
evaluations: Array<{
time: string;
version: number;
provider: string;
model_id: string;
summary_title: string;
summary: string;
score: number;
cost: number;
duration_ms: number;
cases: Array<{
case: string;
score: number;
cost: number;
duration_ms: number;
runs: ExampleRun[];
}>;
}>;
} {
return {
evaluations: EXAMPLE_EVALUATIONS.map((e) => {
const cases = e.cases.map((c) => ({
case: c.case,
score: average(c.runs.map((r) => r.score)),
cost: average(c.runs.map((r) => r.cost)),
duration_ms: average(c.runs.map((r) => r.duration_ms)),
runs: c.runs,
}));
return {
time: e.time,
version: e.version,
provider: e.provider,
model_id: e.model_id,
summary_title: e.summary_title,
summary: e.summary,
score: sum(cases.map((c) => c.score)),
cost: sum(cases.map((c) => c.cost)),
duration_ms: sum(cases.map((c) => c.duration_ms)),
cases,
};
}),
};
}
/**
* Provisions the example Benchmark: if `benchmarks/` already exists (the user already has a
* case library), does nothing; otherwise creates `benchmarks/example-benchmark/` (config, the
* two sample cases, and the scoreboard). Callers are restricted to the default_agent
* initialization path (see agent-state.ts).
*/
export async function provisionExampleBenchmark(
root: string,
projectId: string,
agentId: string,
): Promise<void> {
const dir = benchmarksDir(root, projectId, agentId);
try {
await fs.access(dir);
return;
} catch {
// benchmarks/ does not exist: proceed with provisioning.
}
const benchDir = path.join(dir, EXAMPLE_BENCHMARK_ID);
await Promise.all(
EXAMPLE_CASES.flatMap((c) => [
fs.mkdir(path.join(benchDir, c.id, "statement"), { recursive: true }),
fs.mkdir(path.join(benchDir, c.id, "rubric"), { recursive: true }),
]),
);
await Promise.all([
fs.writeFile(
path.join(benchDir, "benchmark_config.toml"),
`${stringifyToml(EXAMPLE_BENCHMARK_CONFIG)}\n`,
"utf8",
),
fs.writeFile(
path.join(benchDir, "scoreboard.yaml"),
stringifyYaml(buildExampleScoreboard()),
"utf8",
),
...EXAMPLE_CASES.flatMap((c) => [
fs.writeFile(path.join(benchDir, c.id, "statement", "README.md"), c.statement, "utf8"),
fs.writeFile(path.join(benchDir, c.id, "rubric", "README.md"), c.rubric, "utf8"),
]),
]);
}
+16
View File
@@ -0,0 +1,16 @@
/**
* Agent State and Project config storage.
*
* Directory layout, default config, Project config read/write, Agent State load/init.
*/
export * from "./paths.js";
export * from "./default-config.js";
export * from "./builtin-agents.js";
export * from "./model-catalog.js";
export * from "./project-config.js";
export * from "./agent-state.js";
export * from "./agent-vault.js";
export * from "./example-benchmark.js";
// Skill library types and frontmatter parser (from the skills package; server reuses the same implementation via core).
export { parseSkillFrontmatter, type SkillMetadata } from "@prismshadow/penguin-skills";
+489
View File
@@ -0,0 +1,489 @@
/**
* Built-in model catalog (single source of truth): official chat models that AgentHub can
* auto-route, shared by core's default config, server's initial config, and web/cli display.
* Data verified as of 2026-07-10.
* Docs: packages/docs/content/models.{zh,en}.md (site path /docs/models) documents the
* provider groups and credential resolution described here.
*
* Three-bucket pricing convention (USD per million tokens, matching usageToTokenCounts'
* token-to-bucket mapping):
* - cache_read: the vendor's "cache hit" price;
* - cache_write: the vendor's "cache write" price (e.g. Anthropic uses 1.25 x input); vendors
* without a separate cache-write fee use the standard input price;
* - output: output price (thinking + reply).
* OpenAI charges extra for >272K input and Gemini 3.1 Pro for >200K input under official
* long-context pricing; this catalog only records the base tier (the cost center uses a
* single rate, so long-context usage will be underestimated).
*
* Scope: excludes deepseek-chat / deepseek-reasoner legacy aliases that AgentHub cannot
* auto-route (deprecated 2026-07-24), glm-5v-turbo (image input unsupported by AgentHub's GLM
* client), non-chat models (embedding / image generation / TTS), and Bedrock plus
* OpenRouter / SiliconFlow gateway mirror ids. Every model id in this catalog can be
* auto-routed by AgentHub via substring matching, so none set client_type; only custom
* OpenAI-protocol models need `client_type: "openai"`.
*
* This file imports no Node built-ins (type-only imports only), so it can be bundled directly
* for the browser.
*/
import type { ModelEntry, ModelPricing } from "./project-config.js";
/** Model provider info (used for web grouping/logo and the "API key blank falls back to env var" hint). */
export interface ModelProviderInfo {
id: string;
/** Display name (brand name, shared by Chinese and English UI). */
label: string;
/** API key env var name (AgentHub reads this automatically when credential is blank). */
envKey: string;
/** base URL env var name. */
envBaseUrlKey: string;
/** Console URL for obtaining an API key (frontend links this in the group header); none for custom. */
apiKeyUrl?: string;
/** Vendor's model list / docs page URL (frontend's "add model" dialog links this as "get model id"); none for custom. */
modelsUrl?: string;
/**
* Gateway's OpenAI-compatible endpoint (openrouter / siliconflow): used by the frontend's
* "add model" dialog to prefill base URL by group; left blank for direct vendors and custom.
*/
gatewayBaseUrl?: string;
}
/** A single built-in model's catalog entry (`modelId` is the upstream id; paired with `provider` it forms the catalog's unique key). */
export interface ModelCatalogEntry {
modelId: string;
displayName: string;
/** Provider id (one of MODEL_PROVIDERS). */
provider: string;
contextWindow?: number;
pricing?: ModelPricing;
/** Whether image input (vision modality) is supported. */
supportsVision: boolean;
/** AgentHub client protocol: required for models whose id can't be auto-routed (e.g. OpenRouter gateway models). */
clientType?: string;
/** Preset base URL (gateway models): inlined into the model entry so the user only needs to supply an API key. */
baseUrl?: string;
}
/** Each gateway's OpenAI-compatible endpoint (preset base URL for gateway models; also used as the provider's gatewayBaseUrl). */
const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
const SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1";
/**
* Provider list (web model page groups in this order): DeepSeek first (the default model's
* provider), followed by the OpenRouter and SiliconFlow gateways, then Google Gemini before
* Anthropic; custom groups custom OpenAI-protocol models and comes last.
*/
export const MODEL_PROVIDERS: ModelProviderInfo[] = [
{
id: "deepseek",
label: "DeepSeek",
envKey: "DEEPSEEK_API_KEY",
envBaseUrlKey: "DEEPSEEK_BASE_URL",
apiKeyUrl: "https://platform.deepseek.com/api_keys",
modelsUrl: "https://api-docs.deepseek.com/quick_start/pricing",
},
// Gateways (their model ids can't be auto-routed by AgentHub, so they always use
// client_type=openai + a preset base URL): they go through AgentHub's OpenAI client, so when
// credential is blank the SDK reads **OPENAI_API_KEY / OPENAI_BASE_URL** (not the provider's
// own var names) - the env fallback hint must reflect that accurately.
{
id: "openrouter",
label: "OpenRouter",
envKey: "OPENAI_API_KEY",
envBaseUrlKey: "OPENAI_BASE_URL",
apiKeyUrl: "https://openrouter.ai/workspaces/default/keys",
modelsUrl: "https://openrouter.ai/models",
gatewayBaseUrl: OPENROUTER_BASE_URL,
},
{
id: "siliconflow",
label: "SiliconFlow",
envKey: "OPENAI_API_KEY",
envBaseUrlKey: "OPENAI_BASE_URL",
apiKeyUrl: "https://cloud.siliconflow.cn/me/account/ak",
modelsUrl: "https://cloud.siliconflow.cn/models",
gatewayBaseUrl: SILICONFLOW_BASE_URL,
},
{
id: "google",
label: "Google Gemini",
envKey: "GEMINI_API_KEY",
envBaseUrlKey: "GEMINI_BASE_URL",
apiKeyUrl: "https://aistudio.google.com/api-keys",
modelsUrl: "https://ai.google.dev/gemini-api/docs/models",
},
{
id: "anthropic",
label: "Anthropic",
envKey: "ANTHROPIC_API_KEY",
envBaseUrlKey: "ANTHROPIC_BASE_URL",
apiKeyUrl: "https://platform.claude.com/settings/keys",
modelsUrl: "https://docs.claude.com/en/docs/about-claude/models/overview",
},
{
id: "openai",
label: "OpenAI",
envKey: "OPENAI_API_KEY",
envBaseUrlKey: "OPENAI_BASE_URL",
apiKeyUrl: "https://platform.openai.com/api-keys",
modelsUrl: "https://platform.openai.com/docs/models",
},
{
id: "zhipu",
label: "Z.AI (GLM)",
envKey: "ZAI_API_KEY",
envBaseUrlKey: "ZAI_BASE_URL",
apiKeyUrl: "https://open.bigmodel.cn/apikey/platform",
modelsUrl: "https://docs.z.ai/guides/overview/pricing",
},
{
id: "moonshot",
label: "Moonshot (Kimi)",
envKey: "MOONSHOT_API_KEY",
envBaseUrlKey: "MOONSHOT_BASE_URL",
apiKeyUrl: "https://platform.kimi.com/console/api-keys",
modelsUrl: "https://platform.kimi.com/docs/pricing",
},
{ id: "custom", label: "Custom", envKey: "OPENAI_API_KEY", envBaseUrlKey: "OPENAI_BASE_URL" },
];
/** Three-bucket price literal (unit fixed to usd_per_mtok). */
/**
* Converts official CNY pricing to USD for storage (prices are always persisted in USD). The
* conversion rate matches the web display's 7:1 convention, so switching the UI to CNY shows
* exactly the vendor's official CNY price.
*/
function cny(cacheRead: number, cacheWrite: number, output: number): ModelPricing {
const r = (v: number): number => Math.round((v / 7) * 1e6) / 1e6;
return usd(r(cacheRead), r(cacheWrite), r(output));
}
function usd(cacheRead: number, cacheWrite: number, output: number): ModelPricing {
return { unit: "usd_per_mtok", cache_read: cacheRead, cache_write: cacheWrite, output };
}
/** Built-in model catalog (clustered by provider; within each provider, ordered by capability/price, highest first). */
export const MODEL_CATALOG: ModelCatalogEntry[] = [
// -- DeepSeek (official CNY pricing: cache hit / cache miss / output) --
{
modelId: "deepseek-v4-pro",
displayName: "DeepSeek V4 Pro",
provider: "deepseek",
contextWindow: 1000000,
pricing: cny(0.025, 3, 6),
supportsVision: false,
},
{
modelId: "deepseek-v4-flash",
displayName: "DeepSeek V4 Flash",
provider: "deepseek",
contextWindow: 1000000,
pricing: cny(0.02, 1, 2),
supportsVision: false,
},
// —— Anthropic ——
{
modelId: "claude-opus-4-8",
displayName: "Claude Opus 4.8",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.5, 6.25, 25),
supportsVision: true,
},
{
modelId: "claude-opus-4-7",
displayName: "Claude Opus 4.7",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.5, 6.25, 25),
supportsVision: true,
},
{
modelId: "claude-sonnet-4-6",
displayName: "Claude Sonnet 4.6",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.3, 3.75, 15),
supportsVision: true,
},
// —— OpenAI ——
{
modelId: "gpt-5.5",
displayName: "GPT-5.5",
provider: "openai",
contextWindow: 1050000,
pricing: usd(0.5, 5, 30),
supportsVision: true,
},
{
// No official cache discount: cache_read uses the standard input price.
modelId: "gpt-5.5-pro",
displayName: "GPT-5.5 Pro",
provider: "openai",
contextWindow: 1050000,
pricing: usd(30, 30, 180),
supportsVision: true,
},
{
modelId: "gpt-5.4",
displayName: "GPT-5.4",
provider: "openai",
contextWindow: 1050000,
pricing: usd(0.25, 2.5, 15),
supportsVision: true,
},
{
modelId: "gpt-5.4-mini",
displayName: "GPT-5.4 mini",
provider: "openai",
contextWindow: 400000,
pricing: usd(0.075, 0.75, 4.5),
supportsVision: true,
},
{
modelId: "gpt-5.4-nano",
displayName: "GPT-5.4 nano",
provider: "openai",
contextWindow: 400000,
pricing: usd(0.02, 0.2, 1.25),
supportsVision: true,
},
{
// No official cache discount: cache_read uses the standard input price.
modelId: "gpt-5.4-pro",
displayName: "GPT-5.4 Pro",
provider: "openai",
contextWindow: 1050000,
pricing: usd(30, 30, 180),
supportsVision: true,
},
// —— Google Gemini ——
{
// ≤200K input tier; >200K has official surcharge pricing (see file header comment).
modelId: "gemini-3.1-pro-preview",
displayName: "Gemini 3.1 Pro (Preview)",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.2, 2, 12),
supportsVision: true,
},
{
modelId: "gemini-3.5-flash",
displayName: "Gemini 3.5 Flash",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.15, 1.5, 9),
supportsVision: true,
},
{
modelId: "gemini-3-flash-preview",
displayName: "Gemini 3 Flash (Preview)",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.05, 0.5, 3),
supportsVision: true,
},
{
modelId: "gemini-3.1-flash-lite",
displayName: "Gemini 3.1 Flash-Lite",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.025, 0.25, 1.5),
supportsVision: true,
},
// —— Z.AI (GLM) ——
{
modelId: "glm-5.2",
displayName: "GLM-5.2",
provider: "zhipu",
contextWindow: 1000000,
pricing: usd(0.26, 1.4, 4.4),
supportsVision: false,
},
{
modelId: "glm-5.1",
displayName: "GLM-5.1",
provider: "zhipu",
contextWindow: 200000,
pricing: usd(0.26, 1.4, 4.4),
supportsVision: false,
},
{
modelId: "glm-5",
displayName: "GLM-5",
provider: "zhipu",
contextWindow: 200000,
pricing: usd(0.2, 1, 3.2),
supportsVision: false,
},
// -- Moonshot (Kimi) (official CNY pricing) --
{
modelId: "kimi-k2.6",
displayName: "Kimi K2.6",
provider: "moonshot",
contextWindow: 262144,
pricing: cny(1.1, 6.5, 27),
supportsVision: true,
},
{
modelId: "kimi-k2.5",
displayName: "Kimi K2.5",
provider: "moonshot",
contextWindow: 262144,
pricing: cny(0.7, 4, 21),
supportsVision: true,
},
// -- OpenRouter (gateway: uses OpenAI-compatible protocol, preset base URL) --
{
modelId: "xiaomi/mimo-v2.5",
displayName: "MiMo-V2.5",
provider: "openrouter",
contextWindow: 1048576,
pricing: usd(0.0028, 0.14, 0.28),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "tencent/hy3",
displayName: "Hy3",
provider: "openrouter",
contextWindow: 262144,
pricing: usd(0.035, 0.14, 0.58),
supportsVision: false,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// No official separate cache price published: cache_read uses the standard input price (no discount assumed).
modelId: "minimax/minimax-m3",
displayName: "MiniMax M3",
provider: "openrouter",
contextWindow: 1048576,
pricing: usd(0.06, 0.3, 1.2),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// No official separate cache price published: cache_read uses the standard input price.
modelId: "stepfun/step-3.7-flash",
displayName: "Step 3.7 Flash",
provider: "openrouter",
contextWindow: 256000,
pricing: usd(0.04, 0.2, 1.15),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
// -- SiliconFlow (gateway, official CNY pricing: cache hit / input / output) --
{
modelId: "zai-org/GLM-5.2",
displayName: "GLM-5.2",
provider: "siliconflow",
contextWindow: 1000000,
pricing: cny(2, 8, 28),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "deepseek-ai/DeepSeek-V4-Pro",
displayName: "DeepSeek V4 Pro",
provider: "siliconflow",
contextWindow: 1000000,
pricing: cny(0.1, 12, 24),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "meituan-longcat/LongCat-2.0",
displayName: "LongCat 2.0",
provider: "siliconflow",
contextWindow: 1000000,
pricing: cny(0.1, 5, 20),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
];
/** Looks up a catalog entry by (provider, upstream id) pair (**the sole catalog-matching entry point**); returns undefined if not in the catalog. */
export function catalogEntryFor(
provider: string,
upstreamId: string,
): ModelCatalogEntry | undefined {
return MODEL_CATALOG.find((m) => m.provider === provider && m.modelId === upstreamId);
}
/**
* Infers the provider for an upstream id from the built-in catalog (used to default
* `provider` on `model add`): if it matches a catalog entry, use that entry's provider
* (upstream ids are globally unique within the catalog); otherwise custom.
*/
export function inferProviderForUpstream(upstreamId: string): string {
return MODEL_CATALOG.find((m) => m.modelId === upstreamId)?.provider ?? "custom";
}
/** Looks up provider info by provider id; returns undefined for an unknown id. */
export function providerInfo(providerId: string): ModelProviderInfo | undefined {
return MODEL_PROVIDERS.find((p) => p.id === providerId);
}
/** Env var fallback for a single model (the var names AgentHub's client actually reads when api_key / base_url is blank). */
export interface ModelEnvInfo {
envKey: string;
envBaseUrlKey: string;
}
/**
* Resolves the env var fallback for a model: mirrors AgentHub's
* AutoLLMClient routing rules (verified against agenthub v0.3.3 autoClient.ts) - an explicit
* client_type takes priority, otherwise routes to a client by lowercase substring match on
* model_id, returning the var pair that client reads; branch order matches AutoLLMClient.
* Returns undefined on no match (AgentHub will reject that id: it needs an explicit
* client_type, or should be added under custom / a self-built group via the OpenAI protocol).
*/
export function resolveModelEnv(modelId: string, clientType?: string): ModelEnvInfo | undefined {
const t = (clientType || modelId).toLowerCase();
const env = (prefix: string): ModelEnvInfo => ({
envKey: `${prefix}_API_KEY`,
envBaseUrlKey: `${prefix}_BASE_URL`,
});
if (t.includes("gemini-3") || t.includes("gemini-embedding")) return env("GEMINI");
if (
t.includes("claude") &&
(t.includes("4-7") || t.includes("4-8") || t.includes("-5") || t.includes("4-6"))
) {
return env("ANTHROPIC");
}
if (t.includes("gpt-5.4") || t.includes("gpt-5.5")) return env("OPENAI");
if (t.includes("glm-5")) return env("ZAI");
if (t.includes("kimi-k2.5") || t.includes("kimi-k2.6")) return env("MOONSHOT");
if (t.includes("deepseek-v4")) return env("DEEPSEEK");
if (t.includes("openai")) return env("OPENAI");
return undefined;
}
/**
* Catalog -> preset ModelEntry list (shared by defaultProjectConfig and the server's initial
* config, avoiding duplicate hand-written copies). `provider` and `model_id` are persisted as
* separate fields (`model_id` is the plain upstream id); models whose upstream id can be
* auto-routed by AgentHub leave client_type unset; gateway models (OpenRouter / SiliconFlow)
* explicitly set client_type=openai and inline a preset base_url (no secrets included, so the
* user only needs to supply an API key).
*/
export function presetModelEntries(): ModelEntry[] {
return MODEL_CATALOG.map((m) => ({
provider: m.provider,
model_id: m.modelId,
...(m.contextWindow !== undefined ? { context_window: m.contextWindow } : {}),
...(m.clientType !== undefined ? { client_type: m.clientType } : {}),
...(m.pricing ? { pricing: { ...m.pricing } } : {}),
// ModelEntry.vision defaults to supported: only models that don't support images
// explicitly persist false (drives the read_image / describe_image choice and input
// image hand-off, see project-config.ts).
...(m.supportsVision ? {} : { vision: false }),
...(m.baseUrl !== undefined ? { base_url: m.baseUrl } : {}),
}));
}
+114
View File
@@ -0,0 +1,114 @@
/**
* Local directory layout for Agent State and Project config.
*
* Strictly follows the `~/.penguin/data/<project>/agents/<agent>/...` structure.
* This module only provides constants and pure path functions; it never creates directories or reads/writes files.
* Docs: /docs/sessions-and-traces § "Data layout".
*/
import os from "node:os";
import path from "node:path";
/** Default Project id used when none is specified. */
export const DEFAULT_PROJECT_ID = "default_project";
/** Default Agent id used when none is specified. */
export const DEFAULT_AGENT_ID = "default_agent";
/**
* Resolves the local data root directory.
* Prefers the `PENGUIN_HOME` environment variable, otherwise falls back to `~/.penguin/data`
* (under the hidden `~/.penguin` home so it never collides with unrelated folders, and in a
* `data/` subdir kept separate from the installer's binaries under `~/.penguin`).
*/
export function resolveRoot(): string {
return process.env.PENGUIN_HOME ?? path.join(os.homedir(), ".penguin", "data");
}
/** `<root>/<projectId>`. */
export function projectDir(root: string, projectId: string): string {
return path.join(root, projectId);
}
/** `<projectDir>/agents`, the container directory holding every Agent in the Project. */
export function agentsDir(root: string, projectId: string): string {
return path.join(projectDir(root, projectId), "agents");
}
/** `<projectDir>/agents/<agentId>`. */
export function agentDir(root: string, projectId: string, agentId: string): string {
return path.join(agentsDir(root, projectId), agentId);
}
/** `<agentDir>/agent_state`. */
export function agentStateDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "agent_state");
}
/** `<agentDir>/traces`. */
export function tracesDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "traces");
}
/** `<agentDir>/scratchpad`, the Agent's temporary/draft file directory (the model creates a subdirectory per Session id). */
export function scratchpadDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "scratchpad");
}
/** `<agentDir>/workspaces`. */
export function workspacesDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "workspaces");
}
/**
* `<projectDir>/.project_config.toml`, the Project's single config file (a hidden file, not
* shown by default `ls`, written with mode 0600; model entries are inlined with their credential,
* see state/project-config.ts).
*/
export function projectConfigPath(root: string, projectId: string): string {
return path.join(projectDir(root, projectId), ".project_config.toml");
}
/** `<agentStateDir>/system_config.yaml`. */
export function systemConfigPath(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "system_config.yaml");
}
/** `<agentStateDir>/AGENTS.md`. */
export function agentsMdPath(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "AGENTS.md");
}
/** `<agentStateDir>/.vault.toml`, the Agent-level environment-variable vault (see state/agent-vault.ts). */
export function agentVaultPath(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), ".vault.toml");
}
/** `<agentStateDir>/tools`, reserved for user-defined Tool config. */
export function toolsDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "tools");
}
/** `<agentStateDir>/memory`. */
export function memoryDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "memory");
}
/** `<agentStateDir>/skills`. */
export function skillsDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "skills");
}
/** `<agentStateDir>/schedule`, the scheduled-task directory (doesn't exist when unconfigured). */
export function scheduleDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "schedule");
}
/** `<agentDir>/benchmarks`, the capability-evaluation question bank and scores (doesn't exist when unconfigured). */
export function benchmarksDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "benchmarks");
}
/** `<agentDir>/snapshots`, Agent State version snapshots (doesn't exist when unconfigured). */
export function snapshotsDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "snapshots");
}
+432
View File
@@ -0,0 +1,432 @@
/**
* Project config storage (`<project>/.project_config.toml`).
*
* Records the available Models, the default Model, and each Model's credential (Model
* is decoupled from Agent — the Model selection isn't stored in Agent State, but maintained by
* the Project). Config is persisted as TOML.
*
* `.project_config.toml` is the Project's **single config file**: a hidden file (not shown by
* default `ls`), written to disk with mode 0600; credentials (api_key / base_url) are **inlined
* on the model entry** rather than split into a supplementary area and a separate secrets file.
* It can only be read/written via the system interfaces (CLI / Web) — never hand-edited by the
* model or the user; the system Prompt is forbidden from reading this file, `loadProjectConfig`
* returns plaintext, and masking is applied at the interface layer (when shown by server / cli).
*
* Model references are **fully split into separate fields**: an entry stores
* `provider` and `model_id` as two independent fields, with the `(provider, model_id)` pair as
* the unique key — string concatenation like `<provider>/<id>` is forbidden anywhere in the
* pipeline. `model_id` is the upstream request id, sent to AgentHub unchanged; `default_model` /
* `vision_model` are paired `{ provider, model_id }` references (a TOML inline table).
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseToml, stringify as stringifyToml } from "smol-toml";
import { inferProviderForUpstream, presetModelEntries } from "./model-catalog.js";
import { projectConfigPath } from "./paths.js";
/** Model reference: a `(provider, model_id)` pair (never string-concatenated anywhere). */
export interface ModelRef {
provider: string;
/** Upstream model id (the request id sent to AgentHub unchanged). */
model_id: string;
}
/**
* Display form of a paired reference (shared by error messages and CLI output):
* `(provider=..., model_id=...)`. For display only — it isn't any storage or addressing format.
*/
export function formatModelRef(ref: ModelRef): string {
return `(provider=${ref.provider}, model_id=${ref.model_id})`;
}
/**
* Pricing for a single Model: three price buckets, in USD per million tokens.
* Docs: /docs/configuration § "Project config".
*/
export interface ModelPricing {
/** Pricing unit tag; currently only `usd_per_mtok` (USD per million tokens). */
unit: "usd_per_mtok";
cache_read: number;
cache_write: number;
output: number;
}
/**
* A single available Model entry (credential inlined, single config file).
* Docs: /docs/models § "The per-Project model table".
*/
export interface ModelEntry {
/** provider group (stored separately from `model_id`; the pair is the entry's unique key). */
provider: string;
/** Upstream model id: the actual request id sent to AgentHub, used paired with provider for display, pricing, and stats. */
model_id: string;
context_window?: number;
/**
* AgentHub client protocol (`openai` / `claude-4-8` / `deepseek-v4` / …); defaults to being
* inferred by AgentHub from the request id (`model_id`). A third-party model speaking the
* OpenAI protocol should set this to `openai`.
*/
client_type?: string;
/**
* Display name (the model page card title): only persisted when it differs from the builtin
* catalog (the user renamed it / a custom model); when not persisted, it's inferred from the
* builtin catalog by `(provider, model_id)`, falling back to displaying model_id if it can't be
* inferred.
*/
display_name?: string;
/**
* Whether image input is supported (vision/multimodal); defaults to supported. For a model
* tagged `false` (e.g. DeepSeek): images from conversation input are saved to the session
* scratchpad and handed over as a file path spliced into the text, and the image-reading tool
* switches to describe_image (a vision model reads on its behalf) — the image never directly
* enters that session's history.
*/
vision?: boolean;
/** Pricing info; absent means this Model's cost isn't counted. */
pricing?: ModelPricing;
/** API key (inlined credential); left empty falls back to the vendor's environment variable. */
api_key?: string;
/** Custom base URL (inlined credential); preset for gateway models. */
base_url?: string;
/** api_key's write timestamp (ISO 8601; a display field maintained by the interface layer). */
created_at?: string;
}
/**
* Project-level config.
* Docs: /docs/configuration § "Project config".
*/
export interface ProjectConfig {
/** Project display name (the display name is separate from the id, shown as the id when unset). */
name?: string;
/** Paired reference to the default Model; must point to an entry in `models`. */
default_model?: ModelRef;
/**
* The vision model used by read_image to read on behalf of a session model (when a session
* model with `vision=false` reads an image, it's handed to this model to describe and the tool
* returns text); must point to an entry in `models` (a paired reference). Unconfigured by
* default — models that don't support images won't be able to read images.
*/
vision_model?: ModelRef;
models: ModelEntry[];
}
/**
* Returns the Project's default config: every entry from the preset builtin model catalog
* (including context_window / pricing / vision tags and the preset base_url for gateway models,
* with no keys included) — the user only needs to fill in an API key as needed (left empty falls
* back to the vendor's environment variable).
*/
export function defaultProjectConfig(): ProjectConfig {
return {
default_model: { provider: "deepseek", model_id: "deepseek-v4-pro" },
models: presetModelEntries(),
};
}
/** The old format (concatenated storage id / string reference) is never migrated: reading it reports a clear error immediately (the product hasn't shipped yet). */
const OLD_FORMAT_HINT =
"产品未发布不做迁移:请删除该配置文件后用 `penguin config model add/default` 重建。";
/** Validates the default_model / vision_model fields: must be a { provider, model_id } paired reference. */
function parseRefField(file: string, name: string, value: unknown): ModelRef | undefined {
if (value === undefined) return undefined;
const ref = value as { provider?: unknown; model_id?: unknown };
if (
typeof value !== "object" ||
value === null ||
typeof ref.provider !== "string" ||
typeof ref.model_id !== "string"
) {
throw new Error(
`.project_config.toml 的 ${name} 是旧版本/非法格式(须为 { provider = "...", model_id = "..." } 成对引用):${file}。${OLD_FORMAT_HINT}`,
);
}
return { provider: ref.provider, model_id: ref.model_id };
}
/** Validates a model entry: both provider and model_id must be strings (an old-format entry is missing provider). */
function assertModelEntry(file: string, entry: unknown): ModelEntry {
const m = entry as { provider?: unknown; model_id?: unknown };
if (
typeof entry !== "object" ||
entry === null ||
typeof m.provider !== "string" ||
typeof m.model_id !== "string"
) {
throw new Error(
`.project_config.toml 的 models 条目是旧版本/非法格式(provider 与 model_id 须为两个独立字段):${file}。${OLD_FORMAT_HINT}`,
);
}
return entry as ModelEntry;
}
/**
* Loads the Project config; returns the default config (without writing to disk) if
* `.project_config.toml` doesn't exist. Returns plaintext (masking is applied at the interface
* layer); reports a clear error when the old format (a string reference / an entry missing
* provider) is read.
*/
export async function loadProjectConfig(root: string, projectId: string): Promise<ProjectConfig> {
const file = projectConfigPath(root, projectId);
let raw: string;
try {
raw = await fs.readFile(file, "utf8");
} catch (err) {
if ((err as NodeJS.ErrnoException).code === "ENOENT") return defaultProjectConfig();
throw err;
}
// Defensive: parseToml may return null/undefined for an empty file, and destructuring it would throw a TypeError.
const parsed = (parseToml(raw) ?? {}) as Record<string, unknown>;
const defaultModel = parseRefField(file, "default_model", parsed.default_model);
const visionModel = parseRefField(file, "vision_model", parsed.vision_model);
return {
...(parsed.name !== undefined ? { name: parsed.name as string } : {}),
...(defaultModel !== undefined ? { default_model: defaultModel } : {}),
...(visionModel !== undefined ? { vision_model: visionModel } : {}),
models: ((parsed.models as unknown[] | undefined) ?? []).map((m) => assertModelEntry(file, m)),
};
}
/** A TOML inline table for a paired reference (reuses smol-toml's string serialization, guaranteeing correct escaping). */
function tomlInlineRef(ref: ModelRef): string {
const kv = (obj: Record<string, string>): string => stringifyToml(obj).trim();
return `{ ${kv({ provider: ref.provider })}, ${kv({ model_id: ref.model_id })} }`;
}
/** Whether a value has the paired-reference shape ({ provider, model_id }, two string fields). */
function isModelRefShape(v: unknown): v is ModelRef {
if (v === null || typeof v !== "object" || Array.isArray(v)) return false;
const o = v as Record<string, unknown>;
return typeof o.provider === "string" && typeof o.model_id === "string";
}
/**
* Renders the full text of `.project_config.toml` — the **single source of the write format
* site-wide** (shared by core's saveProjectConfig and the interface layer's full-table write, to
* avoid the same file ending up in two different formats).
*
* Paired references (default_model / vision_model) are rendered as a TOML inline table
* `{ provider = "...", model_id = "..." }`; `models` is always
* placed last, since any table header after `[[models]]` would be read as its sub-table. Unknown
* extension fields are kept as-is.
*/
export function renderProjectConfigToml(data: Record<string, unknown>): string {
const head: string[] = [];
for (const [key, value] of Object.entries(data)) {
if (value === undefined || key === "models") continue;
head.push(
isModelRefShape(value)
? `${key} = ${tomlInlineRef(value)}`
: stringifyToml({ [key]: value }).trim(),
);
}
const models = Array.isArray(data.models) ? data.models : [];
return [...head, stringifyToml({ models })].join("\n");
}
/**
* Saves the Project config: writes the full table to the single config file
* `.project_config.toml`. The file contains secrets like api_key, so it's written to disk with
* mode 0600 (a hidden file blocks `ls`, not reads; mode only takes effect on creation, so chmod
* converges an existing file too).
*/
export async function saveProjectConfig(
root: string,
projectId: string,
cfg: ProjectConfig,
): Promise<void> {
const file = projectConfigPath(root, projectId);
await fs.mkdir(path.dirname(file), { recursive: true });
await fs.writeFile(file, renderProjectConfigToml({ ...cfg }), {
encoding: "utf8",
mode: 0o600,
});
await fs.chmod(file, 0o600);
}
/**
* Adds or updates a Model:
* - Upserts into `models`, deduplicated by the `(provider, model_id)` pair (provider may be
* omitted — the builtin catalog is used to infer the upstream id's group, falling back to
* custom if it can't be inferred);
* - If `api_key`/`base_url` are provided, they're written inline into the entry;
* - Set as the default Model (a paired reference) when `opts.setDefault` is true.
* Reads the existing config (or the default), saves after the change, and returns the updated
* config.
*/
export async function addModel(
root: string,
projectId: string,
entry: {
/** provider group; inferred from the builtin catalog when omitted (`inferProviderForUpstream`, falling back to custom if it can't be inferred). */
provider?: string;
/** Upstream model id (sent to AgentHub unchanged). */
model_id: string;
context_window?: number;
client_type?: string;
/** Whether image input is supported (vision/multimodal); keeps the existing value by default (treated as supported if never set). */
vision?: boolean;
/** Price input may cover only some buckets; merged and written as a complete `ModelPricing`. */
pricing?: Partial<ModelPricing>;
api_key?: string;
base_url?: string;
},
opts?: { setDefault?: boolean },
): Promise<ProjectConfig> {
const cfg = await loadProjectConfig(root, projectId);
const provider = entry.provider ?? inferProviderForUpstream(entry.model_id);
// upsert: layers new fields on top of the existing entry; fields not explicitly provided
// (e.g. context_window) keep their existing value, so a call like "just add an api_key"
// doesn't wipe out the prior config.
const idx = cfg.models.findIndex((m) => m.provider === provider && m.model_id === entry.model_id);
const existing = idx >= 0 ? cfg.models[idx] : undefined;
const modelEntry: ModelEntry = {
provider,
model_id: entry.model_id,
};
const contextWindow = entry.context_window ?? existing?.context_window;
if (contextWindow !== undefined) {
modelEntry.context_window = contextWindow;
}
const clientType = entry.client_type ?? existing?.client_type;
if (clientType !== undefined) {
modelEntry.client_type = clientType;
}
// The display name and api_key write timestamp are not set by this function; kept as-is on upsert.
if (existing?.display_name !== undefined) {
modelEntry.display_name = existing.display_name;
}
const vision = entry.vision ?? existing?.vision;
if (vision !== undefined) {
modelEntry.vision = vision;
}
// The three price buckets are merged field by field: an unspecified bucket keeps its existing
// value (the same policy as context_window/credential); the unit is fixed to usd_per_mtok, and
// the complete pricing is written as long as any bucket is present.
const mergedPricing: Partial<ModelPricing> = {
...existing?.pricing,
...entry.pricing,
};
if (
mergedPricing.cache_read !== undefined ||
mergedPricing.cache_write !== undefined ||
mergedPricing.output !== undefined
) {
modelEntry.pricing = {
unit: "usd_per_mtok",
cache_read: mergedPricing.cache_read ?? 0,
cache_write: mergedPricing.cache_write ?? 0,
output: mergedPricing.output ?? 0,
};
}
// Inline credential entry: fields not provided keep their existing value.
const apiKey = entry.api_key ?? existing?.api_key;
if (apiKey !== undefined) {
modelEntry.api_key = apiKey;
}
const baseUrl = entry.base_url ?? existing?.base_url;
if (baseUrl !== undefined) {
modelEntry.base_url = baseUrl;
}
if (existing?.created_at !== undefined) {
modelEntry.created_at = existing.created_at;
}
if (idx >= 0) {
cfg.models[idx] = modelEntry;
} else {
cfg.models.push(modelEntry);
}
if (opts?.setDefault) {
cfg.default_model = { provider, model_id: entry.model_id };
}
await saveProjectConfig(root, projectId, cfg);
return cfg;
}
/**
* Sets the default Model and saves. The target reference must exist in `models` (a reference
* pointing outside the config would make createSession error immediately); throws otherwise.
*/
export async function setDefaultModel(
root: string,
projectId: string,
ref: ModelRef,
): Promise<ProjectConfig> {
const cfg = await loadProjectConfig(root, projectId);
if (!getModel(cfg, ref)) {
throw new Error(
`default_model 必须指向已配置的模型:${formatModelRef(ref)} 不在 models 中。用 \`penguin config model list\` 查看已配置的模型。`,
);
}
cfg.default_model = { provider: ref.provider, model_id: ref.model_id };
await saveProjectConfig(root, projectId, cfg);
return cfg;
}
/**
* Sets the vision model used to read images on behalf of read_image, and saves. The target
* reference must exist in `models` and not be tagged `vision=false` (a model that doesn't support
* images can't read on someone's behalf); throws otherwise.
*/
export async function setVisionModel(
root: string,
projectId: string,
ref: ModelRef,
): Promise<ProjectConfig> {
const cfg = await loadProjectConfig(root, projectId);
const entry = getModel(cfg, ref);
if (!entry) {
throw new Error(
`vision_model 必须指向已配置的模型:${formatModelRef(ref)} 不在 models 中。用 \`penguin config model list\` 查看已配置的模型。`,
);
}
if (entry.vision === false) {
throw new Error(`vision_model 不能指向标注为不支持图片的模型:${formatModelRef(ref)}。`);
}
cfg.vision_model = { provider: ref.provider, model_id: ref.model_id };
await saveProjectConfig(root, projectId, cfg);
return cfg;
}
/** Looks up a Model entry exactly by its `(provider, model_id)` paired reference; returns `undefined` if it doesn't exist. */
export function getModel(cfg: ProjectConfig, ref: ModelRef): ModelEntry | undefined {
return cfg.models.find((m) => m.provider === ref.provider && m.model_id === ref.model_id);
}
/**
* Resolves a model reference (the **single entry point for "provider omitted"**, shared by core
* and CLI/server — never set up a second one):
* - `provider` given: validated for existence by exact paired reference;
* - `provider` omitted: an exact-match lookup on `model_id` (no fuzzy matching of any kind) —
* resolvable only when **exactly one** entry matches; 0 or multiple matches always report a
* clear error (an ambiguity error lists the candidate paired references).
*/
export function resolveModelRef(cfg: ProjectConfig, modelId: string, provider?: string): ModelRef {
if (provider !== undefined) {
const ref: ModelRef = { provider, model_id: modelId };
if (!getModel(cfg, ref)) {
throw new Error(
`Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`,
);
}
return ref;
}
const candidates = cfg.models.filter((m) => m.model_id === modelId);
if (candidates.length === 1) {
return { provider: candidates[0]!.provider, model_id: modelId };
}
if (candidates.length === 0) {
throw new Error(
`Model 不在 Project 配置中:没有 model_id 为 ${modelId} 的条目。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`,
);
}
throw new Error(
`模型引用有歧义:model_id ${modelId} 命中多个条目——${candidates
.map((m) => formatModelRef({ provider: m.provider, model_id: m.model_id }))
.join("、")}。请补充 provider 以给出成对引用。`,
);
}
+10
View File
@@ -0,0 +1,10 @@
export { Writer, readTrace } from "./writer.js";
export type { WriterOptions } from "./writer.js";
export {
findLatestTraceFile,
latestSessionId,
parseTraceLines,
readTraceTolerant,
resumeTrace,
} from "./resume.js";
export type { LocatedTraceFile, ResumeResult } from "./resume.js";
+449
View File
@@ -0,0 +1,449 @@
/**
* Trace replay — the core of Session resume.
*
* Replay produces two results: the **history** injected via setHistory (committed turns only),
* and the **carry-over** input resent with the first `run` after resume. Resume is **best-effort**:
* Trace only records real messages, so synthesized carry-over (`<turn_aborted>` flattening, pairing
* placeholders) is never written to Trace — replay reconstructs from the original messages
* (unanswered input is resent as-is, pairing placeholders are resynthesized as needed). History is
* guaranteed to be **structurally valid** (turns complete, tool_call pairs matched), not a
* byte-for-byte match of what AgentHub actually received; incomplete model output (thinking/text)
* is allowed to be lost.
*
* Messages are attributed to a Request by **position**, not by content inspection:
* - Input = user-side messages accumulated after the previous `request` `stop` (the first
* Request is `session_meta`) and before this `start` (messages are written to Trace before
* being sent with the request); user-side messages that land between `start` and `stop`
* (output from parallel tools completing during the request) count toward the **next** turn's
* input.
* - Output = assistant messages between this `start` and `stop`.
*
* Determination order: first check file-level compaction closure; then evaluate turn by turn
* (completed turns go to history, others are dropped wholesale while keeping outputs paired with
* already-committed tool_calls); finally, the remaining input is the carry-over, with pairing
* backfill applied.
* Docs: /docs/sessions-and-traces § "Session recovery".
*/
import { readdir, readFile } from "node:fs/promises";
import { join } from "node:path";
import {
emptyTokenCounts,
isCompleteModelMessage,
isEventMessage,
isSessionMeta,
toolCallOutput,
userText,
} from "../omnimessage/index.js";
import type {
CompactionEndPayload,
CompleteModelMessage,
OmniMessage,
RequestBeginPayload,
RequestEndPayload,
SessionMetaMessage,
TokenCounts,
TokenUsagePayload,
ToolCallPayload,
} from "../omnimessage/index.js";
import { extractSummary } from "../engine/context-engine.js";
/** Replay result: all the state needed to resume a Session. */
export interface ResumeResult {
/** Committed history (complete model_msg, in order), injected in one shot via setHistory; empty on compaction closure. */
history: CompleteModelMessage[];
/**
* Pending input (carry-over): resent alongside new input with the first `run` after resume.
* Already includes pairing-backfill placeholders — placeholders exist only in memory
* (synthesized carry-over is never written to Trace) and are resynthesized on each resume.
*/
carryOver: OmniMessage[];
/** Compaction closure (file-level): this file's context is fully closed; resume starts a new, empty context. */
contextClosed: boolean;
/** Compaction closure in summarize mode: the reconstructed `<context_summary>` summary, prepended to the next run's input. */
pendingSummary?: OmniMessage;
/** Session-level cumulative Token carry-over (the session value from the last token_usage). */
sessionTokens: TokenCounts;
/** The request.total from the last token_usage (context usage figure). */
lastRequestTotal: number;
/** Session cumulative turn count carry-over (count of completed requests). */
sessionTurns: number;
/** Rendering view: this context's complete model_msg plus key event_msg entries (including interrupted turns and their markers); empty on compaction closure. */
renderMessages: OmniMessage[];
/** The file's first session_meta; null if missing (unresumable — the caller reports the error). */
meta: SessionMetaMessage | null;
}
/** Content of the pairing-backfill placeholder output (the tool hadn't finished and no output was persisted before the process exited). */
const PROCESS_EXIT_PLACEHOLDER = "[interrupted: process exited before the tool finished]";
/**
* Parse Trace JSONL content. Tolerates a **truncated last line** left behind by an abnormal
* process exit (that line is ignored); corruption in the middle is outside the crash window
* (append-only, single writer), so it throws loudly.
*/
export function parseTraceLines(content: string): OmniMessage[] {
const lines = content.split("\n");
const out: OmniMessage[] = [];
for (let i = 0; i < lines.length; i++) {
const line = lines[i]!.trim();
if (!line) continue;
try {
out.push(JSON.parse(line) as OmniMessage);
} catch (err) {
const isLastNonEmpty = lines.slice(i + 1).every((l) => l.trim().length === 0);
if (isLastNonEmpty) break;
throw err;
}
}
return out;
}
/** Read and parse a Trace file (tolerates a truncated last line). */
export async function readTraceTolerant(path: string): Promise<OmniMessage[]> {
return parseTraceLines(await readFile(path, "utf8"));
}
/** A located Trace file: its path, containing date-directory name, and index. */
export interface LocatedTraceFile {
path: string;
dateDir: string;
index: number;
}
const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/;
/**
* Locate the **highest-index** Trace file for a Session (one Trace file corresponds to one
* complete model context). Scans `<tracesDir>/<yyyy-mm-dd>/<sessionId>_<index3>.jsonl`; returns
* null if no match is found.
*/
export async function findLatestTraceFile(
tracesDir: string,
sessionId: string,
): Promise<LocatedTraceFile | null> {
let best: LocatedTraceFile | null = null;
for (const dateDir of await listDirs(tracesDir)) {
for (const file of await listFiles(join(tracesDir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (!match || match[1] !== sessionId) continue;
const index = Number(match[2]);
if (!best || index > best.index) {
best = { path: join(tracesDir, dateDir, file), dateDir, index };
}
}
}
return best;
}
/**
* The id of the most recent Session under this Agent, determined by the timestamp embedded in
* session_id (ids are zero-padded, so lexical order equals chronological order). Returns null if
* there are no Sessions.
*/
export async function latestSessionId(tracesDir: string): Promise<string | null> {
const dateDirs = (await listDirs(tracesDir)).sort((a, b) => b.localeCompare(a));
for (const dateDir of dateDirs) {
const files = (await listFiles(join(tracesDir, dateDir))).sort((a, b) => b.localeCompare(a));
for (const file of files) {
const match = TRACE_FILE_RE.exec(file);
if (!match) continue;
if (await hasResumableTraceContent(join(tracesDir, dateDir, file))) {
return match[1]!;
}
}
}
return null;
}
async function hasResumableTraceContent(file: string): Promise<boolean> {
try {
const messages = await readTraceTolerant(file);
return messages.some((msg) => !isSessionMeta(msg));
} catch {
return false;
}
}
async function listDirs(dir: string): Promise<string[]> {
try {
const entries = await readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isDirectory()).map((e) => e.name);
} catch {
return []; // traces directory doesn't exist yet: no Sessions
}
}
async function listFiles(dir: string): Promise<string[]> {
try {
const entries = await readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isFile()).map((e) => e.name);
} catch {
return [];
}
}
function isRequestBegin(msg: OmniMessage): msg is OmniMessage<RequestBeginPayload> {
return isEventMessage(msg) && (msg.payload as { type?: string }).type === "request_begin";
}
function isRequestEnd(msg: OmniMessage): msg is OmniMessage<RequestEndPayload> {
return isEventMessage(msg) && (msg.payload as { type?: string }).type === "request_end";
}
function isCompactionEnd(msg: OmniMessage): msg is OmniMessage<CompactionEndPayload> {
return isEventMessage(msg) && (msg.payload as { type?: string }).type === "compaction_end";
}
function toolCallOutputId(msg: OmniMessage): string | null {
const p = msg.payload as { type?: string; tool_call_id?: string };
return p.type === "tool_call_output" ? (p.tool_call_id ?? null) : null;
}
/**
* Replay a Trace file (the current context), reconstructing history and carry-over input.
* Input is the message sequence parsed by `readTraceTolerant`.
*/
export function resumeTrace(messages: OmniMessage[]): ResumeResult {
const meta = (messages.find(isSessionMeta) as SessionMetaMessage | undefined) ?? null;
const sessionTokens = lastSessionTokens(messages);
const lastRequestTotal = lastRequestTotalOf(messages);
// —— First check the file-level case: compaction closure (the file ends with a completed
// compaction stop and no new file was opened, i.e. it's still the latest index at resume time)
// — this file's context is fully closed, so the whole file is not replayed.
const last = messages[messages.length - 1];
if (last && isCompactionEnd(last)) {
const p = last.payload;
if (p.status === "completed") {
const result: ResumeResult = {
history: [],
carryOver: [],
contextClosed: true,
sessionTokens,
lastRequestTotal: 0, // new context has no usage yet
sessionTurns: 0, // turn count resets after compaction completes
renderMessages: [],
meta,
};
if (p.mode === "summarize") {
// Reconstruct the summary from the compaction request's output (the assistant text of
// the last completed Request).
const summaryText = lastCompletedRequestText(messages);
result.pendingSummary = userText(
`<context_summary>\n${extractSummary(summaryText)}\n</context_summary>`,
);
}
return result;
}
}
// —— Turn-by-turn determination + pending-input convergence.
const history: CompleteModelMessage[] = [];
/** Pending-input buffer: user-side messages not yet sent with any committed Request. */
let pending: OmniMessage[] = [];
/** The current Request's input snapshot (frozen at begin) and its outputs. */
let snapshot: OmniMessage[] = [];
let outputs: CompleteModelMessage[] = [];
let inRequest = false;
/** Whether we're between a matched pair of compaction events: the compaction prompt in this
* span is not conversational input and must not be resent as-is if uncommitted. */
let inCompaction = false;
/** Ids of tool_calls that are committed (in history) and ids of outputs that are paired (in
* history input). */
const committedCallIds = new Set<string>();
const answeredIds = new Set<string>();
let sessionTurns = 0;
const renderMessages: OmniMessage[] = [];
const placeholderFor = (id: string): CompleteModelMessage =>
toolCallOutput({
output: PROCESS_EXIT_PLACEHOLDER,
toolCallId: id,
stopReason: "aborted",
}) as CompleteModelMessage;
const dropUncommittedRound = (): void => {
// Uncommitted turn: the whole turn is excluded from history. Its **original input** (user
// text/images and structured tool output) goes back into the pending buffer as-is — best
// effort to resend "the last input that got no response"; incomplete model output
// (thinking/text) is allowed to be lost.
// Exception: a failed compaction turn's compaction prompt is not conversational input and is
// not reclaimed (structured output is still reclaimed, subject to eligibility filtering).
const keep = inCompaction ? snapshot.filter((m) => toolCallOutputId(m) !== null) : snapshot;
pending = [...keep, ...pending];
snapshot = [];
outputs = [];
inRequest = false;
};
for (const msg of messages) {
if (isSessionMeta(msg)) continue;
if (isRequestBegin(msg)) {
// Defensive: the previous turn had begin but no end (shouldn't happen mid-file — a process
// exit only affects the tail) — treat it as uncommitted.
if (inRequest) dropUncommittedRound();
inRequest = true;
snapshot = pending;
pending = [];
continue;
}
if (isRequestEnd(msg)) {
if (msg.payload.status === "completed") {
// Structural eligibility filter (same rule as the final carry-over): a tool_call_output
// in the snapshot is kept only if it pairs with a tool_call that is **committed and not
// yet answered** — tool output dispatched by a dropped turn gets persisted in the next
// turn's input span, but its tool_call isn't in history, so keeping it as-is would create
// an orphan tool_result with no preceding tool_use, which every request would be rejected
// for by the provider after resume ("Replay Rules"' strict pairing guarantee).
const eligible = snapshot.filter((m) => {
const id = toolCallOutputId(m);
return id === null || (committedCallIds.has(id) && !answeredIds.has(id));
});
// Structural repair (best-effort): a tool_call committed earlier but still unpaired — its
// matching output was once sent as synthesized carry-over but **never written to Trace**
// — resynthesize a placeholder before this turn's input, to keep the injected history
// structurally valid (every assistant tool_use is followed by a user tool_result).
const snapshotOutputIds = new Set(
eligible.map(toolCallOutputId).filter((id): id is string => id !== null),
);
for (const id of committedCallIds) {
if (answeredIds.has(id) || snapshotOutputIds.has(id)) continue;
history.push(placeholderFor(id));
answeredIds.add(id);
}
history.push(...(eligible as CompleteModelMessage[]), ...outputs);
for (const id of snapshotOutputIds) answeredIds.add(id);
for (const m of outputs) {
const p = m.payload as Partial<ToolCallPayload>;
if (p.type === "tool_call" && p.stop_reason === "completed" && p.tool_call_id) {
committedCallIds.add(p.tool_call_id);
}
}
sessionTurns += 1;
snapshot = [];
outputs = [];
inRequest = false;
} else {
dropUncommittedRound();
}
continue;
}
if (isCompleteModelMessage(msg)) {
renderMessages.push(msg);
const role = (msg.payload as { role?: string }).role;
if (role === "user") {
// All user-side messages go into the pending buffer: ones between begin/end (output from
// parallel tools completing during the request) count toward the next turn's input.
pending.push(msg);
} else if (inRequest) {
outputs.push(msg);
}
// Defensive: assistant messages outside a span (shouldn't happen) are excluded from
// history, kept only for rendering.
continue;
}
if (isEventMessage(msg)) {
const t = (msg.payload as { type?: string }).type;
if (t === "compaction_begin") inCompaction = true;
else if (t === "compaction_end") inCompaction = false;
else if (t === "abort" && (msg.origin?.length ?? 0) === 0) renderMessages.push(msg);
// Other events (token_usage / approval_decision, etc.) don't participate in turn
// determination.
}
}
// File ends mid-request (begin but no end — the process exited during a request): treat as an
// uncommitted turn.
if (inRequest) dropUncommittedRound();
// —— Structured resend eligibility filter: a tool_call_output in the pending input is kept
// only if it pairs with a tool_call that is **committed and not yet answered**. Tool output
// dispatched by an uncommitted turn itself (which may be persisted before or after that turn's
// end — the tool and LLM streams run concurrently, so finishing order is unpredictable) is
// dropped entirely: its tool_call isn't in history, so a structured resend would form an orphan
// tool_result with no preceding tool_use, which the provider would reject.
pending = pending.filter((m) => {
const id = toolCallOutputId(m);
return id === null || (committedCallIds.has(id) && !answeredIds.has(id));
});
// —— Pairing backfill: for any tool_call committed in history with no matching output in
// either history or pending input, add an interrupted-state placeholder to pending input.
// Placeholders exist only in memory (synthesized carry-over is never written to Trace) and are
// resynthesized on each resume as needed.
const pairedIds = new Set<string>(answeredIds);
for (const m of pending) {
const id = toolCallOutputId(m);
if (id !== null) pairedIds.add(id);
}
const pairingBackfill: OmniMessage[] = [];
for (const id of committedCallIds) {
if (pairedIds.has(id)) continue;
pairingBackfill.push(placeholderFor(id));
}
return {
history,
carryOver: [...pending, ...pairingBackfill],
contextClosed: false,
sessionTokens,
lastRequestTotal,
sessionTurns,
renderMessages,
meta,
};
}
/** The session cumulative total from the last token_usage (zero if none). */
function lastSessionTokens(messages: OmniMessage[]): TokenCounts {
for (let i = messages.length - 1; i >= 0; i--) {
const msg = messages[i]!;
if (!isEventMessage(msg)) continue;
const p = msg.payload as Partial<TokenUsagePayload>;
if (p.type === "token_usage" && p.session) return p.session;
}
return emptyTokenCounts();
}
/** The request.total from the last token_usage (context usage figure; 0 if none). */
function lastRequestTotalOf(messages: OmniMessage[]): number {
for (let i = messages.length - 1; i >= 0; i--) {
const msg = messages[i]!;
if (!isEventMessage(msg)) continue;
const p = msg.payload as Partial<TokenUsagePayload>;
if (p.type === "token_usage" && p.request) return p.request.total;
}
return 0;
}
/** Concatenated assistant text of the last completed Request (on compaction closure, this is the compaction request's output). */
function lastCompletedRequestText(messages: OmniMessage[]): string {
let text = "";
let current = "";
let inRequest = false;
for (const msg of messages) {
if (isRequestBegin(msg)) {
inRequest = true;
current = "";
continue;
}
if (isRequestEnd(msg)) {
{
// A completed request with empty text still overwrites (we take the text of “the last
// completed request”, even if empty) — otherwise a textless compaction output would fall
// back to an earlier turn's normal reply and get mistakenly injected as the summary; the
// in-process path yields an empty summary here (extractSummary(“”)).
if (msg.payload.status === "completed") text = current;
inRequest = false;
}
continue;
}
if (!inRequest || !isCompleteModelMessage(msg)) continue;
const p = msg.payload as { type?: string; role?: string; text?: string };
if (p.type === "text" && p.role === "assistant" && p.text) current += p.text;
}
return text;
}
+149
View File
@@ -0,0 +1,149 @@
/**
* Trace writer — append-only JSON Lines.
*
* Docs: packages/docs/content/sessions-and-traces.{zh,en}.md (site path
* /docs/sessions-and-traces) documents the file layout and recording rules.
*
* Design points:
* - Every observable action is appended to Trace; historical events are never modified in place
* (append-only).
* - One Trace file corresponds to one complete model context; when the context is compacted
* and a new segment is produced, `rotate()` starts a new, separately numbered file.
* - Only "recordable" messages are written: `session_meta`, complete `model_msg`, and all
* `event_msg`; streaming `partial_*` messages are skipped (the producer appends the
* corresponding complete message once the segment ends); nested child-session messages are
* never written (their spawn location is recorded via the `subagent` pointer event written by
* context_engine).
* - Path convention: `<tracesDir>/<yyyy-mm-dd>/<sessionId>_<index3>.jsonl`.
*/
import { appendFile, mkdir, readFile } from "node:fs/promises";
import { dirname, join } from "node:path";
import {
PartialAggregator,
isCompleteModelMessage,
isEventMessage,
isSessionMeta,
} from "../omnimessage/index.js";
import type { OmniMessage } from "../omnimessage/index.js";
import { formatLocalDate } from "../internal/dates.js";
export interface WriterOptions {
/** Trace root directory, typically `<agent>/traces`. */
tracesDir: string;
/** Current Session id, written into the file name. */
sessionId: string;
/** The time used to derive the date subdirectory; defaults to `new Date()`. */
date?: Date;
/**
* Directly specifies the date subdirectory name (used when Session resumption continues
* writing to the original file: the Trace file follows the context, not the date); takes
* priority over `date`.
*/
dateDir?: string;
/** Starting Trace index (used when Session resumption continues the original index); defaults to 1. */
startIndex?: number;
}
/** Zero-pads a Trace index to 3 digits, e.g. 1 -> "001". */
function formatIndex(index: number): string {
return index.toString().padStart(3, "0");
}
/**
* Determines whether an OmniMessage should be written to Trace (skips streaming partial_* and nested child-session messages).
*
* Child-session messages are never written to this Trace: the child Session has its own complete
* Trace, and recording it again would distort this Trace's statistics. The spawn location is
* recorded via the `subagent` pointer event (recording only the child Session id) that
* context_engine writes at the spawn site; when the session is reopened, the server uses this to
* re-attach the child session to its corresponding run_subagent tool card.
* Docs: /docs/sessions-and-traces § "Trace design".
*/
function isRecordable(msg: OmniMessage): boolean {
if (msg.origin && msg.origin.length > 0) return false;
return isCompleteModelMessage(msg) || isEventMessage(msg) || isSessionMeta(msg);
}
/**
* append-only JSONL Trace writer.
*
* Single-writer scenario (MVP): concurrency safety isn't required, but every write uses
* `appendFile` (O_APPEND) rather than caching a file handle and seeking to write, avoiding
* overwriting existing content; this also removes the need for an explicit close.
*/
export class Writer {
private readonly tracesDir: string;
private readonly sessionId: string;
private readonly dateDir: string;
/** Current Trace index, starting at 1; incremented by `rotate()`. */
private index = 1;
/** Set true once the date directory has been created for the current file, to avoid a redundant mkdir. */
private ensuredDirForIndex = -1;
constructor(opts: WriterOptions) {
this.tracesDir = opts.tracesDir;
this.sessionId = opts.sessionId;
this.dateDir = opts.dateDir ?? formatLocalDate(opts.date ?? new Date());
this.index = opts.startIndex ?? 1;
}
/** Absolute path of the current Trace file. */
currentPath(): string {
const fileName = `${this.sessionId}_${formatIndex(this.index)}.jsonl`;
return join(this.tracesDir, this.dateDir, fileName);
}
/**
* Appends one message. Only written if it's a recordable message; streaming `partial_*` is
* skipped. `mkdir -p`s the date directory on the first write to the current file.
*/
async write(msg: OmniMessage): Promise<void> {
if (!isRecordable(msg)) return;
const path = this.currentPath();
if (this.ensuredDirForIndex !== this.index) {
await mkdir(dirname(path), { recursive: true });
this.ensuredDirForIndex = this.index;
}
await appendFile(path, `${JSON.stringify(msg)}\n`, "utf8");
}
/** Writes multiple messages in sequence. */
async writeAll(msgs: OmniMessage[]): Promise<void> {
for (const msg of msgs) {
await this.write(msg);
}
}
/**
* Aggregates a message stream mixed with streaming `partial_*` into complete messages first,
* then writes them per the `write` convention. A convenience helper: `write` skips partial_*
* by default (the producer will already append the complete message), so this method is only
* needed when reconstructing a complete context from raw streaming fragments.
*/
async aggregateAndWrite(msgs: OmniMessage[]): Promise<void> {
const agg = new PartialAggregator();
for (const msg of msgs) {
await this.writeAll(agg.push(msg));
}
await this.writeAll(agg.flush());
}
/**
* Starts a new Trace file: increments the index, so the next `write` goes to the new file.
* Used to split into a separate file when the context is compacted and a new context segment is produced.
* Docs: /docs/sessions-and-traces § "Trace design".
*/
async rotate(): Promise<void> {
this.index += 1;
}
}
/** Parses a Trace file line by line (ignoring blank lines), for testing and later reads. */
export async function readTrace(path: string): Promise<OmniMessage[]> {
const content = await readFile(path, "utf8");
return content
.split("\n")
.filter((line) => line.trim().length > 0)
.map((line) => JSON.parse(line) as OmniMessage);
}