Initialize repository with harness code and assets

Initial import of all source code, config, and README assets: the
packages workspace (cli, core, server, web, docs, landing, skills),
build scripts, tooling config, and CI workflows.

Includes the data-layout revision made on this branch: the local data
root defaults to ~/.penguin/data (PENGUIN_HOME still overrides; the
installer keeps its binaries in ~/.penguin), and every Agent lives
under <project>/agents/<agent>/ — path helpers, the three
agent-enumeration scans, the system prompt, built-in Skills, tests
and docs all follow the new layout.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018ihk8iQuo3kv2aPjAYEPuR
This commit is contained in:
Yaowei Zheng
2026-07-19 14:06:53 +08:00
committed by GitHub
parent 056bed7aeb
commit 45bfae6e94
543 changed files with 92949 additions and 0 deletions
+39
View File
@@ -0,0 +1,39 @@
# @prismshadow/penguin-cli
The PenguinHarness command line. Installs the `penguin` command: an interactive REPL, a one-shot task runner, model / vault configuration, and the launcher for the Web service.
```bash
npm install -g @prismshadow/penguin-cli # requires Node >= 24
```
```bash
penguin web # start the Web service and open http://127.0.0.1:7364
penguin server # same service, headless
penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default
penguin run -m "Create hello.txt containing Hello, Penguin" # one Task, then exit
penguin chat # REPL: /compact, /exit, Ctrl-C interrupts
penguin chat --resume # resume the latest session
```
Tool calls go through an approval gate — `--approve allow-all` (default) `| deny-all | read-only | always-ask`. Data lives under `~/.penguin/data` (`PENGUIN_HOME` or `--root` override); model credentials come from the Project config or provider env vars (e.g. `DEEPSEEK_API_KEY`).
Prefer a one-line install with a bundled Node runtime? See the [installation guide](https://prism-shadow.github.io/penguin-harness/docs/installation).
## Documentation
- [Quickstart](https://prism-shadow.github.io/penguin-harness/docs/quickstart)
- [CLI Reference](https://prism-shadow.github.io/penguin-harness/docs/cli)
- [Configuration Reference](https://prism-shadow.github.io/penguin-harness/docs/configuration)
## Development
```bash
pnpm penguin <args> # run from source (repo root, via tsx)
pnpm --filter @prismshadow/penguin-cli build # tsup → dist/index.js (the penguin bin)
pnpm --filter @prismshadow/penguin-cli typecheck
pnpm --filter @prismshadow/penguin-cli test
```
Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0
+48
View File
@@ -0,0 +1,48 @@
{
"name": "@prismshadow/penguin-cli",
"version": "0.0.1",
"type": "module",
"description": "PenguinHarness CLI: interactive REPL and single-task runner over @prismshadow/penguin-core.",
"license": "Apache-2.0",
"repository": {
"type": "git",
"url": "git+https://github.com/Prism-Shadow/penguin-harness.git",
"directory": "packages/cli"
},
"bin": {
"penguin": "./dist/index.js"
},
"engines": {
"node": ">=24"
},
"scripts": {
"typecheck": "tsc --noEmit -p tsconfig.json",
"test": "vitest run --passWithNoTests",
"build": "tsup",
"penguin": "tsx src/index.ts"
},
"dependencies": {
"@prismshadow/agenthub": "^0.3.3",
"@prismshadow/penguin-core": "workspace:*",
"@prismshadow/penguin-server": "workspace:*",
"@prismshadow/penguin-skills": "workspace:*",
"commander": "^13.0.0",
"dotenv": "^17.0.0",
"smol-toml": "^1.3.0",
"yaml": "^2.5.0"
},
"devDependencies": {
"@types/node": "^24.0.0",
"tsup": "^8.3.0",
"tsx": "^4.20.0",
"typescript": "^5.6.0",
"vitest": "^2.1.0"
},
"files": [
"dist",
"LICENSE"
],
"publishConfig": {
"access": "public"
}
}
+142
View File
@@ -0,0 +1,142 @@
/**
* CLI tool-call approval.
*
* The CLI consumes the output stream of `session.run()`; within a turn, the engine invokes the
* injected `approve` callback for each tool_call.
* Docs: /docs/cli § "Approval modes (--approve)"; /docs/tools § "Approval".
*/
import { createInterface } from "node:readline";
import type { ApprovalDecision, ApproveFn } from "@prismshadow/penguin-core";
import { defaultMessages } from "./i18n.js";
import type { Messages } from "./i18n.js";
/** Valid string values for the `--approve` option (includes the default allow-all, so scripts can specify it explicitly and get the default behavior). */
const APPROVE_MODES = ["allow-all", "deny-all", "read-only", "always-ask"] as const;
/**
* Approval mode (derived from APPROVE_MODES, the single source of truth):
* - `allow-all`: auto-approve every tool (default mode);
* - `deny-all`: auto-reject every tool;
* - `read-only`: auto-approve read-only tools (permission === "r"), defer the rest to a human;
* - `always-ask`: interactive approval for each call.
*/
export type ApprovalMode = (typeof APPROVE_MODES)[number];
/**
* Resolve the approval mode from the CLI: read `--approve`, default `allow-all`; print a
* message and exit if the value is invalid.
*/
export function resolveApprovalMode(approve: string | undefined, t: Messages): ApprovalMode {
if (approve === undefined) return "allow-all";
const v = approve.trim().toLowerCase();
if ((APPROVE_MODES as readonly string[]).includes(v)) {
return v as ApprovalMode;
}
process.stderr.write(`${t.approveModeInvalid(approve)}\n`);
process.exit(1);
}
/**
* Build the `approve` callback for a given permission mode. `toolPermission` looks up a tool's
* permission level; `interactivePrompt` is the actual Q&A used when deferring to a human (run
* uses a one-off prompt, chat uses a persistent readline). Rendering the approval result is not
* done here — `context_engine` emits the decision as an `approval_decision` event, rendered by
* the frontend (see render.ts).
*/
export function makeApprove(args: {
mode: ApprovalMode;
toolPermission: (name: string) => "r" | "rw" | undefined;
interactivePrompt: ApproveFn;
}): ApproveFn {
const { mode, toolPermission, interactivePrompt } = args;
return async (toolCall) => {
const name = toolCall.payload.name;
switch (mode) {
case "allow-all":
return "allow";
case "deny-all":
return "deny";
case "read-only":
// Auto-approve read-only tools; defer read-write/unknown tools to a human.
if (toolPermission(name) === "r") return "allow";
return interactivePrompt(toolCall);
case "always-ask":
default:
return interactivePrompt(toolCall);
}
};
}
export interface PromptApprovalOptions {
/** Message set; resolved from the env var by default. */
t?: Messages;
/** Stream to read the approval answer from; defaults to `process.stdin`. */
input?: NodeJS.ReadableStream;
/** Stream to print the approval prompt to; defaults to `process.stdout`. */
output?: NodeJS.WritableStream;
}
/**
* Callback that rejects the pending approval while `promptApproval` is waiting; `null`
* otherwise. `penguin run` calls `denyActivePrompt()` from a single global SIGINT handler:
* Ctrl-C during approval collapses to "deny" (consistent with chat), and only interrupts the
* whole turn at other times. SIGINT is registered in exactly one place (run); promptApproval no
* longer attaches its own listener.
*/
let activePromptDeny: (() => void) | null = null;
export function denyActivePrompt(): boolean {
if (!activePromptDeny) return false;
activePromptDeny();
return true;
}
/**
* One-off interactive approval Q&A (for non-persistent REPL scenarios like `run`). The pending
* tool call has already been streamed above. Input-stream EOF/close is treated as a deny;
* Ctrl-C while waiting is collapsed to a deny by the caller (run) via `denyActivePrompt`. The
* readline instance is closed after reading.
*/
export function promptApproval(opts: PromptApprovalOptions = {}): Promise<ApprovalDecision> {
const input = opts.input ?? process.stdin;
const output = opts.output ?? process.stdout;
const t = opts.t ?? defaultMessages();
const rl = createInterface({ input, output });
return new Promise<ApprovalDecision>((resolve) => {
let resolved = false;
const finish = (decision: ApprovalDecision) => {
if (resolved) return;
resolved = true;
// Only clear our own slot: even under concurrent prompts (upstream already serializes
// this; this is a defensive check), don't clobber someone else's deny hook.
if (activePromptDeny === deny) activePromptDeny = null;
rl.close();
resolve(decision);
};
const deny = () => finish("deny");
// Ctrl-C during approval is turned into a "deny" via run's global SIGINT calling
// denyActivePrompt (no duplicate SIGINT listener registered here); input-stream EOF/close
// is likewise treated as a deny, to avoid hanging.
activePromptDeny = deny;
rl.on("close", () => finish("deny"));
rl.question(t.approvePrompt(), (answer) => {
// Tool approval defaults to allow: a bare Enter (empty input) counts as allow.
finish(parseApprovalAnswer(answer, "allow"));
});
});
}
/**
* Parse an approval/confirmation answer (trimmed, case-insensitive): `y`/`yes` → allow,
* `n`/`no` → deny, everything else (including empty input/bare Enter) → `fallback`. Tool
* approval defaults to allow (pass `"allow"`); exit/restart-style confirmations default to no
* (the default `"deny"`).
*/
export function parseApprovalAnswer(
answer: string,
fallback: ApprovalDecision = "deny",
): ApprovalDecision {
const normalized = answer.trim().toLowerCase();
if (normalized === "y" || normalized === "yes") return "allow";
if (normalized === "n" || normalized === "no") return "deny";
return fallback;
}
+369
View File
@@ -0,0 +1,369 @@
/**
* `penguin chat` — interactive REPL.
*
* penguin chat [--model-id <id>] [--provider <group>] [--project-id <id>] [--agent-id <id>]
* [--workspace <path>] [--approve <allow-all|deny-all|read-only|always-ask>]
*
* Each line of input starts one conversation turn; `/compact` proactively compacts the
* context (reason=manual); `/exit` or `/quit` exits.
* Uses the current directory when no Workspace is specified.
*
* Multi-line input: trailing `\` continues the line; when the terminal supports bracketed
* paste, a multi-line paste is treated as a single message (sent on Enter).
*
* Ctrl-C behavior (state-dependent): buffer has content -> clear it;
* awaiting approval -> deny; running -> abort the current turn and return to input;
* empty buffer -> show a y/N exit confirmation.
*
* Implementation notes: on a TTY, stdin is put into raw mode with bracketed paste enabled;
* stdin is piped through PasteFilter into a readline created with `terminal: true` — Ctrl-C
* is captured in-process by readline as 'SIGINT' (it never escapes as an OS signal killing
* the process group), and pasted content is held whole by PasteFilter (not split into
* multiple submits by embedded newlines).
* Docs: /docs/cli § "penguin chat".
*/
import { createInterface, type Interface } from "node:readline";
import type { Command } from "commander";
import { createAgent, userText } from "@prismshadow/penguin-core";
import type { ApprovalDecision, OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
import { StreamRenderer, dim, renderHistory } from "../render.js";
import { runTask } from "../task-loop.js";
import { parseApprovalAnswer, resolveApprovalMode } from "../approval.js";
import { LineComposer, PasteFilter } from "../input.js";
import type { Messages } from "../i18n.js";
export type ChatState = "idle" | "running" | "approving" | "confirming-exit";
export type SigintAction = "deny" | "abort" | "clear" | "confirm-exit" | "exit";
/** Pure decision: current state + whether the input buffer is non-empty -> the action Ctrl-C should perform. */
export function decideSigint(state: ChatState, hasBufferedInput: boolean): SigintAction {
if (state === "approving") return "deny";
if (state === "running") return "abort";
if (state === "confirming-exit") return "exit";
return hasBufferedInput ? "clear" : "confirm-exit";
}
interface RlInternals {
line: string;
cursor: number;
_refreshLine?: () => void;
}
const MAIN_PROMPT = "> ";
const CONT_PROMPT = "… ";
export function registerChatCommand(program: Command, t: Messages): void {
program
.command("chat")
.description(t.chat.desc)
.option("--model-id <id>", t.common.modelId)
.option("--provider <group>", t.common.provider)
.option("--project-id <id>", t.common.projectId)
.option("--agent-id <id>", t.common.agentId)
.option("--workspace <path>", t.common.workspace)
.option("--approve <mode>", t.common.approve)
.option("--resume [sessionId]", t.chat.resume)
.action(async (opts) => {
const mode = resolveApprovalMode(opts.approve, t);
const out = process.stdout;
const agent = await createAgent({
...(opts.agentId ? { agentId: opts.agentId } : {}),
...(opts.projectId ? { projectId: opts.projectId } : {}),
});
// --resume: resumes an existing Session. Workspace and
// Model follow the original Session and cannot be overridden; when omitted, resumes
// the current Agent's most recent Session.
let session;
if (opts.resume !== undefined) {
if (opts.workspace || opts.modelId || opts.provider) {
out.write(`${t.error(t.resumeNoOverride())}\n`);
process.exitCode = 1;
return;
}
const sessionId =
typeof opts.resume === "string" ? opts.resume : await agent.latestSessionId();
if (!sessionId) {
out.write(`${t.error(t.resumeNoSession())}\n`);
process.exitCode = 1;
return;
}
session = await agent.resumeSession({ sessionId });
} else {
session = await agent.createSession({
workspaceDir: opts.workspace ?? process.cwd(),
...(opts.modelId ? { modelId: opts.modelId } : {}),
...(opts.provider ? { provider: opts.provider } : {}),
});
}
const renderer = new StreamRenderer(out, t);
out.write(
`${t.header("chat", agent.state.agentId, session.workspaceDir, session.modelId)}\n` +
`${t.chatHints()}\n`,
);
// On resume, first render the history messages of the current context per Trace
// (full messages, including interrupted turns and their markers), then proceed to
// regular input.
if (session.resumedHistory) {
out.write(`${t.resumedBanner(session.sessionId, session.resumedHistory.length)}\n`);
renderHistory(session.resumedHistory, out);
}
// TTY: raw mode + bracketed paste + PasteFilter; non-TTY (pipe/test): read stdin directly.
const isTTY = Boolean(process.stdin.isTTY);
let pasteFilter: PasteFilter | null = null;
let inputStream: NodeJS.ReadableStream = process.stdin;
if (isTTY) {
process.stdin.setRawMode(true);
out.write("\x1b[?2004h");
pasteFilter = new PasteFilter();
process.stdin.pipe(pasteFilter);
inputStream = pasteFilter;
}
const rl = createInterface({
input: inputStream,
output: out,
terminal: isTTY,
});
const rli = rl as unknown as RlInternals;
const composer = new LineComposer();
let state: ChatState = "idle";
let closed = false;
let taskAbort: AbortController | null = null;
let pendingLine: ((line: string | null) => void) | null = null;
let pendingApproval: ((decision: ApprovalDecision) => void) | null = null;
const cleanup = () => {
if (!isTTY) return;
try {
out.write("\x1b[?2004l");
} catch {
/* ignore */
}
try {
process.stdin.setRawMode(false);
} catch {
/* ignore */
}
try {
if (pasteFilter) process.stdin.unpipe(pasteFilter);
} catch {
/* ignore */
}
try {
process.stdin.pause();
} catch {
/* ignore */
}
};
process.once("exit", cleanup);
if (pasteFilter) {
pasteFilter.on("paste", (text: string) => {
if (state !== "idle") return; // ignore paste while running
const { lineCount, normalized } = composer.pushPaste(text);
if (lineCount === 0) return;
out.write(`${normalized}\n`);
rl.setPrompt(CONT_PROMPT);
rl.prompt();
});
}
rl.on("line", (line) => {
if (state === "confirming-exit") {
if (parseApprovalAnswer(line) === "allow") {
rl.close();
} else {
state = "idle";
composer.reset();
out.write("\n");
rl.setPrompt(MAIN_PROMPT);
rl.prompt();
}
return;
}
if (state === "idle" && pendingLine) {
const { message } = composer.pushTypedLine(line);
if (message === undefined) {
// Continuation: show the continuation prompt and keep waiting.
rl.setPrompt(CONT_PROMPT);
rl.prompt();
} else {
const resolve = pendingLine;
pendingLine = null;
resolve(message);
}
return;
}
if (state === "approving" && pendingApproval) {
const resolve = pendingApproval;
pendingApproval = null;
// Tool approval defaults to allow: pressing Enter (empty input) is treated as allow.
resolve(parseApprovalAnswer(line, "allow"));
}
// running: ignore any line typed at this moment.
});
rl.on("SIGINT", () => {
const hasBuffer = rli.line.length > 0 || composer.hasPending();
const action = decideSigint(state, hasBuffer);
if (action === "deny") {
if (pendingApproval) {
const resolve = pendingApproval;
pendingApproval = null;
out.write("\n");
resolve("deny");
}
} else if (action === "abort") {
if (taskAbort && !taskAbort.signal.aborted) {
out.write(`\n${t.taskInterrupted()}\n`);
taskAbort.abort();
}
} else if (action === "clear") {
composer.reset();
rl.setPrompt(MAIN_PROMPT);
clearCurrentLine(rl, rli, out);
} else if (action === "confirm-exit") {
state = "confirming-exit";
rli.line = "";
rli.cursor = 0;
out.write("\n");
rl.setPrompt(t.confirmExit());
rl.prompt();
} else {
out.write("\n");
rl.close();
}
});
rl.on("close", () => {
closed = true;
if (pendingLine) {
const resolve = pendingLine;
pendingLine = null;
resolve(null);
}
});
const askLine = (): Promise<string | null> =>
new Promise((resolve) => {
if (closed) {
resolve(null);
return;
}
state = "idle";
pendingLine = resolve;
composer.reset();
rli.line = "";
rli.cursor = 0;
out.write("\n");
rl.setPrompt(MAIN_PROMPT);
rl.prompt();
});
// Interactive approval prompt: reuses the persistent readline, prompt text is
// localized; the tool call is already rendered above via streaming, so it is not
// re-rendered here.
const interactivePrompt = (_tc: OmniMessage<ToolCallPayload>): Promise<ApprovalDecision> =>
new Promise((resolve) => {
state = "approving";
pendingApproval = (decision) => {
state = "running";
resolve(decision);
};
rl.setPrompt(t.approvePrompt());
rl.prompt();
});
// Whether this Session already has a resumable Trace record: a resumed Session
// naturally has one; a new Session gets one starting from its first Task / compact
// (session_meta is written along with it). This decides whether to print the resume
// command example on exit.
let resumable = opts.resume !== undefined;
try {
for (;;) {
const line = await askLine();
if (line === null) break;
const text = line.trim();
if (text === "/exit" || text === "/quit") break;
if (text.length === 0) continue;
state = "running";
taskAbort = new AbortController();
try {
if (text === "/compact") {
// Proactive context compaction (Task boundary, reason=manual): the renderer
// prints compaction progress; Ctrl-C aborts the compaction via signal
// (preserving the original context). When there's nothing to compact (session
// just started / two consecutive /compact calls), the engine silently returns
// and we add one line of feedback here. Afterwards, settle the renderer's
// counters (endCompact) — compaction usage is already shown on the completion
// line and must not be counted again toward the next task's stats delta.
const startedAt = Date.now();
let sawMessage = false;
try {
for await (const msg of session.compact({
signal: taskAbort.signal,
})) {
sawMessage = true;
resumable = true;
renderer.handle(msg);
}
} finally {
renderer.endCompact(Date.now() - startedAt);
}
if (!sawMessage) out.write(`${t.compactNothing()}\n`);
} else {
resumable = true;
await runTask(session, [userText(text)], {
mode,
signal: taskAbort.signal,
renderer,
interactivePrompt,
t,
});
}
} catch (err) {
out.write(`\n${t.error(err instanceof Error ? err.message : String(err))}\n`);
} finally {
taskAbort = null;
state = "idle";
}
}
} finally {
rl.close();
cleanup();
session.dispose(); // tear down managed long-running command sessions to avoid leaking background processes
process.removeListener("exit", cleanup);
// On exit, print a dimmed resume command example: includes this
// session's Project / Agent options so the command can be copy-pasted directly;
// skipped when the Session has no Trace record yet (nothing to resume).
if (resumable) {
const command =
`penguin chat --resume ${session.sessionId}` +
(opts.projectId ? ` --project-id ${opts.projectId}` : "") +
(opts.agentId ? ` --agent-id ${opts.agentId}` : "");
out.write(`${dim(t.resumeHint(command))}\n`);
}
}
});
}
/** Clear the current input line and redraw the prompt (Ctrl-C clears the buffer when it has content). */
function clearCurrentLine(rl: Interface, rli: RlInternals, out: NodeJS.WritableStream): void {
rli.line = "";
rli.cursor = 0;
if (typeof rli._refreshLine === "function") {
rli._refreshLine();
} else {
out.write("\r\x1b[K");
rl.prompt(true);
}
}
+368
View File
@@ -0,0 +1,368 @@
/**
* `penguin config` — manages a Project's model credentials, default model, model list,
* Agent-level vault environment variables, and UI language.
*
* penguin config model add --model-id <upstream id> [--provider <group>] [--api-key <key>] [--context-window <n>] [--set-default] [--root <dir>]
* penguin config model default --model-id <upstream id> --provider <group> [--root <dir>]
* penguin config model vision --model-id <upstream id> --provider <group> [--root <dir>]
* penguin config model list [--root <dir>]
* penguin config vault set --key <name> --value <value> [--agent-id <id>] [--root <dir>]
* penguin config vault list [--agent-id <id>] [--root <dir>]
* penguin config vault remove --key <name> [--agent-id <id>] [--root <dir>]
* penguin config lang <en|zh>
*
* `--model-id` always takes the **upstream id** (the request id sent to AgentHub verbatim),
* which together with `--provider` forms a `(provider, model_id)` paired reference —
* **no string concatenation is ever performed**. For `model add`, --provider defaults to
* an inference from the built-in catalog (falling back to custom when inference fails);
* a new entry's client_type defaults according to the group's semantics (not set for
* first-party vendors; openai for custom / self-hosted groups / gateways, with the
* gateway's endpoint base URL pre-filled). For `model default` / `model vision`,
* --provider is **required**; core validation raises an error when the reference is not
* found in models. `--root` specifies the data root directory (priority: option >
* PENGUIN_HOME > ~/.penguin/data). The UI language is controlled by the PENGUIN_LANG
* environment variable; `config lang` writes it into the shell startup file and restarts
* the shell to take effect.
* Docs: /docs/cli § "penguin config".
*/
import { homedir } from "node:os";
import path from "node:path";
import { createInterface } from "node:readline";
import type { Command } from "commander";
import {
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
type ModelPricing,
type ModelRef,
type ProjectConfig,
addModel,
catalogEntryFor,
formatModelRef,
getModel,
inferProviderForUpstream,
loadAgentVault,
loadProjectConfig,
providerInfo,
removeVaultEntry,
resolveRoot,
setDefaultModel,
setVaultEntry,
setVisionModel,
} from "@prismshadow/penguin-core";
import { parseApprovalAnswer } from "../approval.js";
import { getMessages, maskApiKey, type Messages } from "../i18n.js";
import { applyLanguageToRc, restartShell } from "../lang-config.js";
/** Data root directory: the `--root` option takes priority (relative paths resolved against cwd), then PENGUIN_HOME / ~/.penguin/data. */
function resolveRootOption(root: string | undefined): string {
return root !== undefined ? path.resolve(root) : resolveRoot();
}
/**
* Renders the model list as column-aligned lines (the default model is marked with `*`;
* fully empty columns are omitted automatically). `provider` and `model_id` each occupy
* their own column (stored fields, never split apart); `vision` reflects the effective
* semantics (the TOML `vision` annotation takes priority, falling back to the catalog
* annotation — matched by the (provider, model_id) pair — and recorded as Y under
* "default = supported" when neither is present). Exported for unit tests.
*/
export function formatModelRows(cfg: ProjectConfig): string[] {
const cells = cfg.models.map((entry) => {
const cat = catalogEntryFor(entry.provider, entry.model_id);
const vision = entry.vision ?? cat?.supportsVision ?? true;
const isDefault =
cfg.default_model?.provider === entry.provider &&
cfg.default_model?.model_id === entry.model_id;
return {
provider: `${isDefault ? "* " : " "}${entry.provider}`,
model: entry.model_id,
vision: `vision=${vision ? "Y" : "-"}`,
context_window:
entry.context_window !== undefined ? `context_window=${entry.context_window}` : "",
client_type: entry.client_type ? `client_type=${entry.client_type}` : "",
pricing: entry.pricing
? `price=${entry.pricing.cache_read}/${entry.pricing.cache_write}/${entry.pricing.output}`
: "",
api_key: `api_key=${maskApiKey(entry.api_key)}`,
base_url: entry.base_url ? `base_url=${entry.base_url}` : "",
};
});
const columns = [
"provider",
"model",
"vision",
"context_window",
"client_type",
"pricing",
"api_key",
"base_url",
] as const;
const widths = columns.map((c) => Math.max(...cells.map((cell) => cell[c].length)));
const active = columns
.map((c, i) => ({ key: c, width: widths[i]! }))
.filter((col) => col.width > 0);
return cells.map((cell) =>
active
.map((col, i) => (i === active.length - 1 ? cell[col.key] : cell[col.key].padEnd(col.width)))
.join(" ")
.trimEnd(),
);
}
export function registerConfigCommand(program: Command, t: Messages): void {
const config = program.command("config").description(t.config.desc);
const model = config.command("model").description(t.config.modelDesc);
model
.command("add")
.description(t.config.addDesc)
.requiredOption("--model-id <id>", t.config.addModelId)
.option("--provider <group>", t.config.addProvider)
.option("--api-key <key>", t.config.addApiKey)
.option("--base-url <url>", t.config.addBaseUrl)
.option("--context-window <n>", t.config.addContextWindow, parseIntArg)
.option("--client-type <type>", t.config.addClientType)
// Tri-state: --vision marks it supported / --no-vision marks it unsupported / neither given keeps the existing value (defaults to supported).
.option("--vision", t.config.addVision)
.option("--no-vision", t.config.addNoVision)
.option("--price-cache-read <n>", t.config.addPriceCacheRead, parseFloatArg)
.option("--price-cache-write <n>", t.config.addPriceCacheWrite, parseFloatArg)
.option("--price-output <n>", t.config.addPriceOutput, parseFloatArg)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--set-default", t.config.addSetDefault, false)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
// --model-id takes the upstream id, paired with --provider as a reference
// (--provider defaults to catalog-based inference, falling back to custom); no
// concatenation is performed.
const modelId: string = opts.modelId;
const provider: string = opts.provider ?? inferProviderForUpstream(modelId);
const ref: ModelRef = { provider, model_id: modelId };
const before = await loadProjectConfig(root, opts.projectId);
const existed = getModel(before, ref) !== undefined;
// client_type default rule, only injected for new entries (updating an
// existing entry never overrides an explicit config): not set for first-party
// vendor groups (AgentHub auto-routes by upstream id, with env fallback keyed on
// id); defaults to openai for custom / self-hosted / gateway groups, with the
// gateway's endpoint base URL pre-filled as well.
const pInfo = providerInfo(provider);
const openAiDefault =
pInfo === undefined || pInfo.id === "custom" || pInfo.gatewayBaseUrl !== undefined;
const clientType: string | undefined =
opts.clientType ?? (!existed && openAiDefault ? "openai" : undefined);
const baseUrl: string | undefined =
opts.baseUrl ?? (!existed ? pInfo?.gatewayBaseUrl : undefined);
// Only collect explicitly given price fields, letting addModel merge them with the existing pricing per-field.
const pricing: Partial<ModelPricing> = {};
if (opts.priceCacheRead !== undefined) pricing.cache_read = opts.priceCacheRead;
if (opts.priceCacheWrite !== undefined) pricing.cache_write = opts.priceCacheWrite;
if (opts.priceOutput !== undefined) pricing.output = opts.priceOutput;
const cfg = await addModel(
root,
opts.projectId,
{
provider,
model_id: modelId,
...(opts.contextWindow !== undefined ? { context_window: opts.contextWindow } : {}),
...(clientType !== undefined ? { client_type: clientType } : {}),
...(opts.vision !== undefined ? { vision: opts.vision } : {}),
...(Object.keys(pricing).length > 0 ? { pricing } : {}),
...(opts.apiKey !== undefined ? { api_key: opts.apiKey } : {}),
...(baseUrl !== undefined ? { base_url: baseUrl } : {}),
},
{ setDefault: Boolean(opts.setDefault) },
);
const defaultRef = cfg.default_model && formatModelRef(cfg.default_model);
const line = existed
? t.modelUpdated(formatModelRef(ref), defaultRef)
: t.modelAdded(formatModelRef(ref), defaultRef);
process.stdout.write(`${line}\n`);
});
model
.command("default")
.description(t.config.defaultDesc)
.requiredOption("--model-id <id>", t.config.refModelId)
.requiredOption("--provider <group>", t.config.refProvider)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
// --model-id takes the upstream id, paired with the required --provider as a
// reference (no concatenation, no fuzzy matching); setDefaultModel raises an error
// when the reference is not found in models.
const ref: ModelRef = { provider: opts.provider, model_id: opts.modelId };
try {
await setDefaultModel(root, opts.projectId, ref);
} catch (err) {
process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`);
process.exitCode = 1;
return;
}
process.stdout.write(`${t.defaultModelSet(formatModelRef(ref))}\n`);
});
model
.command("vision")
.description(t.config.visionDesc)
.requiredOption("--model-id <id>", t.config.refModelId)
.requiredOption("--provider <group>", t.config.refProvider)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
// Paired reference semantics match `model default`; existence and vision=false semantics validation is handled by setVisionModel.
const ref: ModelRef = { provider: opts.provider, model_id: opts.modelId };
try {
await setVisionModel(root, opts.projectId, ref);
} catch (err) {
process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`);
process.exitCode = 1;
return;
}
process.stdout.write(`${t.visionModelSet(formatModelRef(ref))}\n`);
});
model
.command("list")
.description(t.config.listDesc)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
const cfg = await loadProjectConfig(root, opts.projectId);
if (cfg.models.length === 0) {
process.stdout.write(`${t.modelListEmpty()}\n`);
return;
}
process.stdout.write(`${t.modelListTitle()}\n`);
for (const line of formatModelRows(cfg)) {
process.stdout.write(`${line}\n`);
}
});
const vault = config.command("vault").description(t.config.vaultDesc);
vault
.command("set")
.description(t.config.vaultSetDesc)
.requiredOption("--key <name>", t.config.vaultKey)
.requiredOption("--value <value>", t.config.vaultValue)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--agent-id <id>", t.common.agentId, DEFAULT_AGENT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
try {
await setVaultEntry(root, opts.projectId, opts.agentId, opts.key, opts.value);
} catch (err) {
// Validation errors such as an invalid key name: print an explanation and exit with a non-zero code, without throwing a stack trace.
process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`);
process.exitCode = 1;
return;
}
process.stdout.write(`${t.vaultSet(opts.key)}\n`);
});
vault
.command("list")
.description(t.config.vaultListDesc)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--agent-id <id>", t.common.agentId, DEFAULT_AGENT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
const entries = Object.entries(await loadAgentVault(root, opts.projectId, opts.agentId));
if (entries.length === 0) {
process.stdout.write(`${t.vaultListEmpty()}\n`);
return;
}
process.stdout.write(`${t.vaultListTitle()}\n`);
const width = Math.max(...entries.map(([key]) => key.length));
for (const [key, value] of entries) {
process.stdout.write(`${key.padEnd(width)} ${maskApiKey(value)}\n`);
}
});
vault
.command("remove")
.description(t.config.vaultRemoveDesc)
.requiredOption("--key <name>", t.config.vaultKey)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--agent-id <id>", t.common.agentId, DEFAULT_AGENT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
const vaultEntries = await loadAgentVault(root, opts.projectId, opts.agentId);
if (vaultEntries[opts.key] === undefined) {
process.stderr.write(`${t.vaultKeyMissing(opts.key)}\n`);
process.exitCode = 1;
return;
}
await removeVaultEntry(root, opts.projectId, opts.agentId, opts.key);
process.stdout.write(`${t.vaultRemoved(opts.key)}\n`);
});
config
.command("lang")
.description(t.config.langDesc)
.argument("<language>", t.config.langArg)
.action(async (language: string) => {
const lang = String(language).trim().toLowerCase();
if (lang !== "zh" && lang !== "en") {
process.stderr.write(`${t.langInvalid(String(language))}\n`);
process.exitCode = 1;
return;
}
const { rcPath } = await applyLanguageToRc(lang, {
shell: process.env.SHELL,
home: homedir(),
});
// The confirmation message is shown in the target language; the user must confirm before the shell restarts.
const m = getMessages(lang);
process.stdout.write(`${m.langSet(lang, rcPath)}\n`);
const interactive = Boolean(process.stdin.isTTY && process.stdout.isTTY);
if (interactive && (await confirmYes(m.langRestartConfirm()))) {
process.stdout.write(`${m.langRestart()}\n`);
restartShell(lang);
} else {
process.stdout.write(`${m.langRestartHint(rcPath)}\n`);
}
});
}
/** Interactive y/N confirmation; Ctrl-C (SIGINT) or input stream EOF/close are both treated as no, to avoid hanging. */
function confirmYes(prompt: string): Promise<boolean> {
const rl = createInterface({ input: process.stdin, output: process.stdout });
return new Promise<boolean>((resolve) => {
let done = false;
const finish = (value: boolean) => {
if (done) return;
done = true;
process.off("SIGINT", onSigint);
rl.close();
resolve(value);
};
const onSigint = () => finish(false);
process.once("SIGINT", onSigint);
rl.on("close", () => finish(false));
rl.question(prompt, (answer) => finish(parseApprovalAnswer(answer) === "allow"));
});
}
function parseIntArg(value: string): number {
const n = Number.parseInt(value, 10);
if (Number.isNaN(n)) {
throw new Error(`无效的整数:${value}`);
}
return n;
}
function parseFloatArg(value: string): number {
const n = Number.parseFloat(value);
if (Number.isNaN(n)) {
throw new Error(`无效的数值:${value}`);
}
return n;
}
+76
View File
@@ -0,0 +1,76 @@
/**
* `penguin run` — send a single Task in one shot.
*
* penguin run -m <msg> [--model-id <id>] [--provider <group>] [--workspace <path>]
* [--project-id <id>] [--agent-id <id>]
* [--approve <allow-all|deny-all|read-only|always-ask>]
*
* Uses the current directory when Workspace is unspecified; uses the Project's default model
* when model is unspecified. `--provider` is optional: when omitted, `--model-id` is resolved
* via resolveModelRef semantics (only matches when the exact value is globally unique in the
* config; ambiguity is an error). Defaults to interactive per-call approval; `--approve`
* selects the permission mode.
* Docs: /docs/cli § "penguin run".
*/
import type { Command } from "commander";
import { createAgent, userText } from "@prismshadow/penguin-core";
import { StreamRenderer } from "../render.js";
import { runTask } from "../task-loop.js";
import { denyActivePrompt, resolveApprovalMode } from "../approval.js";
import type { Messages } from "../i18n.js";
export function registerRunCommand(program: Command, t: Messages): void {
program
.command("run")
.description(t.run.desc)
.requiredOption("-m, --message <message>", t.run.message)
.option("--model-id <id>", t.common.modelId)
.option("--provider <group>", t.common.provider)
.option("--project-id <id>", t.common.projectId)
.option("--agent-id <id>", t.common.agentId)
.option("--workspace <path>", t.common.workspace)
.option("--approve <mode>", t.common.approve)
.action(async (opts) => {
const mode = resolveApprovalMode(opts.approve, t);
const agent = await createAgent({
...(opts.agentId ? { agentId: opts.agentId } : {}),
...(opts.projectId ? { projectId: opts.projectId } : {}),
});
const session = await agent.createSession({
workspaceDir: opts.workspace ?? process.cwd(),
...(opts.modelId ? { modelId: opts.modelId } : {}),
...(opts.provider ? { provider: opts.provider } : {}),
});
const out = process.stdout;
out.write(`${t.header("run", agent.state.agentId, session.workspaceDir, session.modelId)}\n`);
const controller = new AbortController();
const onSigint = () => {
// Single SIGINT handler: Ctrl-C during approval collapses to "deny this tool" (see
// approval.ts); at all other times it interrupts the whole turn.
if (denyActivePrompt()) return;
controller.abort();
};
process.on("SIGINT", onSigint);
const renderer = new StreamRenderer(out, t);
try {
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
});
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
// exit code, for scripts/CI to check.
if (result.aborted) process.exitCode = 1;
} finally {
process.off("SIGINT", onSigint);
session.dispose(); // Tear down managed long-running command sessions to avoid leaking background processes
}
out.write("\n");
});
}
+133
View File
@@ -0,0 +1,133 @@
/**
* `penguin server` / `penguin web` — starts the Web service.
*
* penguin server [--port <port>] [--host <host>]
* penguin web [--port <port>] [--host <host>] [--no-open]
*
* Both are entry points into the same service process: after setting PORT / HOST, it
* dynamically imports `@prismshadow/penguin-server` (whose entry point handles dotenv
* loading and graceful shutdown on its own), so the two never listen on separate ports
* in parallel. Port/host priority: command-line option > existing environment variable
* (including .env) > default 7364 / 127.0.0.1. `penguin web` additionally polls until the
* service is ready, prints the URL, and opens a browser per-platform (`--no-open`
* disables this).
* Docs: /docs/cli § "penguin server / penguin web".
*/
import { spawn } from "node:child_process";
import type { Command } from "commander";
import type { Messages } from "../i18n.js";
/** Default service port (deliberately avoids common defaults like 3000/8080). */
export const DEFAULT_PORT = 7364;
/** Default service listen host. */
export const DEFAULT_HOST = "127.0.0.1";
/**
* Resolves the listen port: command-line option takes priority, then the PORT
* environment variable, defaulting to 7364; throws if not an integer or out of the
* 0-65535 range. Exported for unit tests.
*/
export function resolvePort(option: string | undefined, env: string | undefined): number {
const raw = option ?? env;
if (raw === undefined || raw === "") return DEFAULT_PORT;
const port = Number(raw);
if (!Number.isInteger(port) || port < 0 || port > 65535) {
throw new Error(`Invalid port "${raw}". Use an integer between 0 and 65535.`);
}
return port;
}
/**
* Picks the command to open a browser per-platform. On win32, `start` treats the first
* quoted argument as the window title, so an extra empty title placeholder is passed.
* Exported for unit tests.
*/
export function browserCommand(platform: string, url: string): { command: string; args: string[] } {
if (platform === "darwin") return { command: "open", args: [url] };
if (platform === "win32") return { command: "cmd", args: ["/c", "start", "", url] };
return { command: "xdg-open", args: [url] };
}
/** URL used for the readiness probe and browser access: when listening on a wildcard address (0.0.0.0 / ::), access via 127.0.0.1 instead. Exported for unit tests. */
export function browserUrl(host: string, port: number): string {
const target = host === "0.0.0.0" || host === "::" ? "127.0.0.1" : host;
return `http://${target}:${port}/`;
}
/**
* Sets PORT / HOST then starts the service: the server entry point only reads
* process.env, and its dotenv loading never overrides existing environment variables,
* so the values written here are the ones that take effect (options take priority over
* .env and any pre-existing env vars).
*/
async function startServer(opts: {
port?: string;
host?: string;
}): Promise<{ host: string; port: number }> {
const port = resolvePort(opts.port, process.env.PORT);
const host = opts.host ?? process.env.HOST ?? DEFAULT_HOST;
process.env.PORT = String(port);
process.env.HOST = host;
await import("@prismshadow/penguin-server");
return { host, port };
}
/** Polls the service root path until it responds (any HTTP response counts as ready); keeps waiting on connection failure, returns false on timeout. */
async function waitForReady(url: string, timeoutMs = 15_000, intervalMs = 300): Promise<boolean> {
const deadline = Date.now() + timeoutMs;
for (;;) {
try {
// Each probe is capped at 1s: if the port is held by a non-HTTP program, the
// connection can succeed while the response hangs forever; without a timeout this
// would block the whole polling loop (the deadline check below would never run).
const res = await fetch(url, { signal: AbortSignal.timeout(1000) });
void res.body?.cancel();
return true;
} catch {
// The service isn't listening yet (or this probe timed out): keep polling.
}
if (Date.now() >= deadline) return false;
await new Promise((resolve) => setTimeout(resolve, intervalMs));
}
}
/** Opens the browser: spawn detached with output ignored; any failure is silently swallowed (failing to open doesn't affect the running service). */
function openBrowser(url: string): void {
const { command, args } = browserCommand(process.platform, url);
try {
const child = spawn(command, args, { detached: true, stdio: "ignore" });
child.on("error", () => {});
child.unref();
} catch {
// e.g. the browser command doesn't exist: ignore, the user can open it manually.
}
}
export function registerServeCommands(program: Command, t: Messages): void {
program
.command("server")
.description(t.serve.serverDesc)
.option("--port <port>", t.serve.port)
.option("--host <host>", t.serve.host)
.action(async (opts: { port?: string; host?: string }) => {
await startServer(opts);
});
program
.command("web")
.description(t.serve.webDesc)
.option("--port <port>", t.serve.port)
.option("--host <host>", t.serve.host)
.option("--no-open", t.serve.noOpen)
.action(async (opts: { port?: string; host?: string; open: boolean }) => {
const { host, port } = await startServer(opts);
const url = browserUrl(host, port);
const ready = await waitForReady(url);
if (!ready) {
process.stdout.write(`${t.webTimeout(url)}\n`);
return;
}
process.stdout.write(`${t.webReady(url)}\n`);
if (opts.open) openBrowser(url);
});
}
+390
View File
@@ -0,0 +1,390 @@
/**
* CLI text internationalization (i18n).
*
* Language comes from the `PENGUIN_LANG` env var (`en` / `zh`), defaulting to English (en) —
* independent of Project config or CLI options. This module centralizes all user-visible text:
* command/option help descriptions and runtime output, one implementation per language.
*/
/** UI language. */
export type Language = "en" | "zh";
/** Resolve the language from the env var; `zh` matches exactly, everything else falls back to English (see comment #2). */
export function resolveLanguage(): Language {
const v = (process.env.PENGUIN_LANG ?? "").trim().toLowerCase();
return v === "zh" ? "zh" : "en";
}
export interface Messages {
// —— Command/option help descriptions ——
cliDescription: string;
versionDesc: string;
common: {
projectId: string;
agentId: string;
modelId: string;
/** run/chat's --provider: pairs with --model-id; when omitted, resolved by unique match (ambiguity is an error). */
provider: string;
/** Data root directory option (priority: --root > PENGUIN_HOME > ~/.penguin/data). */
root: string;
workspace: string;
approve: string;
};
config: {
desc: string;
modelDesc: string;
addDesc: string;
addModelId: string;
addProvider: string;
addApiKey: string;
addBaseUrl: string;
addContextWindow: string;
addClientType: string;
addVision: string;
addNoVision: string;
addPriceCacheRead: string;
addPriceCacheWrite: string;
addPriceOutput: string;
addSetDefault: string;
defaultDesc: string;
visionDesc: string;
/** `model default` / `model vision`'s --model-id: the upstream request id (pairs with --provider as a reference). */
refModelId: string;
/** `model default` / `model vision`'s --provider: the provider group of the referenced entry (required). */
refProvider: string;
listDesc: string;
langDesc: string;
langArg: string;
vaultDesc: string;
vaultSetDesc: string;
vaultListDesc: string;
vaultRemoveDesc: string;
vaultKey: string;
vaultValue: string;
};
run: { desc: string; message: string };
chat: { desc: string; resume: string };
serve: {
serverDesc: string;
webDesc: string;
port: string;
host: string;
noOpen: string;
};
// —— Runtime output ——
header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string;
chatHints(): string;
confirmExit(): string;
taskInterrupted(): string;
error(message: string): string;
/** Approval prompt text (the tool call is already streamed above and directly precedes this prompt, so no index and no re-rendering). */
approvePrompt(): string;
/**
* Stats shown at the end of each Task: Session cumulative values plus this task's delta —
* context window length, Token usage, elapsed time. Delta strings carry their own sign
* (contextDelta can be negative after context is compacted), e.g.
* `[stats] context 4k (+1k) · tokens 6k (+1.2k) · 5.1s (+2.3s)`.
*/
taskStats(s: {
context: string;
contextDelta: string;
tokens: string;
tokensDelta: string;
elapsed: string;
elapsedDelta: string;
}): string;
/** Abort event label (may include a reason). */
abortLabel(reason?: string): string;
/** request_end ended with timeout/malformed: the engine retries (reconnect) carrying already-produced content; attempt is the retry count. */
reconnectLabel(status: "timeout" | "malformed", attempt: number): string;
/** compaction start event: indicates compaction in progress (mode is summarize/discard, reason is context/turns/manual). */
compactionStart(mode: string, reason: string): string;
/**
* compaction stop event: the compaction result (status is completed/failed/aborted;
* completed varies its text by mode). tokens is Token usage (same convention as the stats
* line: total = Session cumulative, delta = consumed by this compaction, carrying its own
* sign); when present it is appended at the end of the line, e.g. ` · tokens 14k (+6k)`.
*/
compactionStop(mode: string, status: string, tokens?: { total: string; delta: string }): string;
/** Prompt shown when `/compact` has nothing to compact (session just started / two consecutive compactions). */
compactNothing(): string;
/** Prompt for an invalid --approve mode. */
approveModeInvalid(value: string): string;
/** Render label for an approval decision (frontend renders the approval_decision event; one label each for allow/deny). */
approvalDecision(decision: "allow" | "deny"): string;
/** --resume is mutually exclusive with --workspace/--model-id (neither can change once the Session is created). */
resumeNoOverride(): string;
/** --resume given without a session id, and the current Agent has no Session at all. */
resumeNoSession(): string;
/** One-line prompt shown after a successful resume, before rendering history. */
resumedBanner(sessionId: string, messageCount: number): string;
/** Example resume command shown when the REPL exits (dim print; only when this session has a resumable record). */
resumeHint(command: string): string;
langInvalid(value: string): string;
langSet(lang: string, rcPath: string): string;
langRestartConfirm(): string;
langRestart(): string;
langRestartHint(rcPath: string): string;
/** Result output for model add/default/vision: the argument is the already-formatted pair reference (formatModelRef). */
modelAdded(model: string, defaultModel: string | undefined): string;
modelUpdated(model: string, defaultModel: string | undefined): string;
defaultModelSet(model: string): string;
visionModelSet(model: string): string;
modelListTitle(): string;
modelListEmpty(): string;
vaultSet(key: string): string;
vaultRemoved(key: string): string;
vaultKeyMissing(key: string): string;
vaultListTitle(): string;
vaultListEmpty(): string;
/** URL prompt once the `penguin web` service is ready. */
webReady(url: string): string;
/** Manual-open prompt after the `penguin web` ready-poll times out (15s). */
webTimeout(url: string): string;
}
function header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string {
return `PenguinHarness ${kind} — agent=${agentId} workspace=${workspace} model=${model}`;
}
const en: Messages = {
cliDescription: "PenguinHarness CLI",
versionDesc: "output the version number",
common: {
projectId: "Project id",
agentId: "Agent id",
modelId: "Model to use (upstream model id; defaults to the Project default model)",
provider:
"Provider of --model-id; when omitted, the model id must match exactly one configured entry (ambiguity is an error)",
root: "Data root directory (overrides PENGUIN_HOME and ~/.penguin/data)",
workspace: "Workspace directory; must already exist (defaults to the current directory)",
approve:
"Approval mode: allow-all (auto-approve, default), deny-all (auto-reject), read-only (auto-approve read-only tools, prompt for the rest), always-ask (prompt per tool)",
},
config: {
desc: "Manage Project configuration",
modelDesc: "Manage model credentials and the default model",
addDesc: "Add or update a model, optionally writing a credential",
addModelId: "Upstream model id sent to AgentHub as-is (e.g. claude-sonnet-4-6)",
addProvider:
"Provider group stored alongside model_id; inferred from the builtin catalog when omitted, else custom",
addApiKey: "API key, stored inline in the Project's hidden .project_config.toml",
addBaseUrl: "Custom base URL",
addContextWindow: "Context window size (tokens)",
addClientType: "AgentHub client type (e.g. openai); inferred from model id when omitted",
addVision: "Mark the model as supporting image input (vision)",
addNoVision: "Mark the model as NOT supporting image input; omit both to keep current",
addPriceCacheRead: "Price per 1M tokens: cache read (USD)",
addPriceCacheWrite: "Price per 1M tokens: cache write (USD)",
addPriceOutput: "Price per 1M tokens: output (USD)",
addSetDefault: "Also set as the Project default model",
defaultDesc: "Set the Project default model",
visionDesc: "Set the vision model used by read_image for non-vision session models",
refModelId: "Upstream model id; forms the (provider, model_id) pair reference with --provider",
refProvider: "Provider group of the referenced entry (see `penguin config model list`)",
listDesc: "List the Project's models (API keys hidden)",
langDesc:
"Set the interface language (en|zh); persists PENGUIN_LANG to your shell startup file",
langArg: "Language: en or zh",
vaultDesc: "Manage an Agent's vault (environment variables injected into its shell commands)",
vaultSetDesc: "Set a vault environment variable (added or overwritten)",
vaultListDesc: "List vault environment variables (values masked)",
vaultRemoveDesc: "Remove a vault environment variable",
vaultKey: "Variable name (letters, digits and underscores; must not start with a digit)",
vaultValue: "Variable value, written to the Agent's agent_state/.vault.toml",
},
run: { desc: "Run a single Task", message: "Prompt for this Task" },
chat: {
desc: "Open the interactive REPL",
resume:
"Resume an existing Session (defaults to the agent's most recent one); workspace and model follow the original Session",
},
serve: {
serverDesc: "Start the Web service (HTTP API and the built-in frontend, same process)",
webDesc: "Start the Web service and open the UI in a browser once it is ready",
port: "Listen port (falls back to the PORT env var, default 7364)",
host: "Listen address (falls back to the HOST env var, default 127.0.0.1)",
noOpen: "Do not open a browser automatically",
},
header,
chatHints: () =>
"Type a message to start a conversation; end a line with \\; /compact to compact the context; /exit to quit; and Ctrl-C interrupts the current conversation.",
confirmExit: () => "Exit penguin? [y/N] ",
taskInterrupted: () => "[current conversation interrupted]",
error: (message) => `[error] ${message}`,
approvePrompt: () => "? Approve this tool call? [Y/n] ",
taskStats: (s) =>
`[stats] context ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · ${s.elapsed} (${s.elapsedDelta})`,
abortLabel: (reason) => `[abort]${reason ? `: ${reason}` : ""}`,
reconnectLabel: (status, attempt) =>
`[retry] ${status === "timeout" ? "connection timed out" : "response incomplete or unparseable"}; sending retry #${attempt}…`,
compactionStart: (mode, reason) =>
mode === "discard"
? `[compaction] discarding context (${reason})…`
: `[compaction] summarizing context (${reason})…`,
compactionStop: (mode, status, tokens) =>
(status === "completed"
? mode === "discard"
? "[compaction] done; old context discarded"
: "[compaction] done; continuing with the summarized context"
: `[compaction] ${status}; keeping the current context`) +
(tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""),
compactNothing: () => "[compaction] nothing to compact yet",
approveModeInvalid: (value) =>
`Invalid approval mode "${value}". Use allow-all, deny-all, read-only, or always-ask.`,
approvalDecision: (decision) => (decision === "allow" ? "✓ [approved]" : "× [denied]"),
resumeNoOverride: () =>
"--resume does not accept --workspace, --model-id or --provider: they follow the original Session and cannot change.",
resumeNoSession: () => "No session to resume: this agent has no recorded sessions yet.",
resumedBanner: (sessionId, messageCount) =>
`[resumed] ${sessionId} · ${messageCount} message${messageCount === 1 ? "" : "s"} in the current context`,
resumeHint: (command) => `To continue this conversation: ${command}`,
langInvalid: (value) => `Invalid language "${value}". Use en or zh.`,
langSet: (lang, rcPath) => `Language set to ${lang}; wrote PENGUIN_LANG to ${rcPath}.`,
langRestartConfirm: () => "Open a new shell now to apply? [y/N] ",
langRestart: () => "Opening a new shell with the new language (type exit to return)…",
langRestartHint: (rcPath) => `Open a new terminal, or run: source ${rcPath}`,
modelAdded: (model, def) => `Added model ${model}. Default model: ${def ?? "(unset)"}`,
modelUpdated: (model, def) => `Updated model ${model}. Default model: ${def ?? "(unset)"}`,
defaultModelSet: (model) => `Default model set to ${model}.`,
visionModelSet: (model) => `Vision model set to ${model}.`,
modelListTitle: () => "Configured models:",
modelListEmpty: () => "No models configured yet. Add one with `penguin config model add`.",
vaultSet: (key) => `Saved vault entry ${key}.`,
vaultRemoved: (key) => `Removed vault entry ${key}.`,
vaultKeyMissing: (key) => `Vault entry ${key} does not exist.`,
vaultListTitle: () => "Vault environment variables (values masked):",
vaultListEmpty: () => "The vault is empty. Add one with `penguin config vault set`.",
webReady: (url) => `Web UI ready: ${url}`,
webTimeout: (url) => `Server is not responding yet; open ${url} manually once it is ready.`,
};
const zh: Messages = {
cliDescription: "PenguinHarness CLI",
versionDesc: "输出版本号",
common: {
projectId: "Project id",
agentId: "Agent id",
modelId: "本次使用的模型(上游模型 id;默认 Project 默认模型)",
provider: "--model-id 的 provider 分组;省略时 model id 须在配置中精确唯一命中(歧义报错)",
root: "数据根目录(优先于 PENGUIN_HOME 与 ~/.penguin/data)",
workspace: "Workspace 目录,须为已存在目录(默认当前目录)",
approve:
"审批模式:allow-all(全部放行,缺省)、deny-all(全部拒绝)、read-only(自动放行只读工具,其余仍逐个询问)、always-ask(逐个询问)",
},
config: {
desc: "管理 Project 配置",
modelDesc: "管理模型 credential 与默认模型",
addDesc: "新增或更新一个模型,并可写入 credential",
addModelId: "上游模型 id(如 claude-sonnet-4-6,原样发给 AgentHub)",
addProvider: "与 model_id 分列存储的 provider 分组;缺省按内置目录推断,推断不出为 custom",
addApiKey: "API key,内联存入 Project 的隐藏文件 .project_config.toml",
addBaseUrl: "自定义 base url",
addContextWindow: "上下文窗口大小(token 数)",
addClientType: "AgentHub 客户端协议(如 openai);缺省由 model id 推断",
addVision: "标注该模型支持图片输入(视觉)",
addNoVision: "标注该模型不支持图片输入;两者都不给则保留原值",
addPriceCacheRead: "每百万 token 价格:缓存读取(USD)",
addPriceCacheWrite: "每百万 token 价格:缓存写入(USD)",
addPriceOutput: "每百万 token 价格:输出(USD)",
addSetDefault: "同时设为该 Project 的默认模型",
defaultDesc: "设置 Project 的默认模型",
visionDesc: "设置 read_image 代读用的视觉模型(供不支持图片的会话模型读图)",
refModelId: "上游模型 id;与 --provider 构成 (provider, model_id) 成对引用",
refProvider: "引用条目的 provider 分组(见 `penguin config model list`)",
listDesc: "列出当前 Project 的模型(API key 隐藏)",
langDesc: "设置界面语言(en|zh);将 PENGUIN_LANG 写入 shell 启动文件并持久化",
langArg: "语言:en 或 zh",
vaultDesc: "管理 Agent vault(注入该 Agent shell 命令的环境变量)",
vaultSetDesc: "写入一个 vault 环境变量(不存在则新增,存在则覆盖)",
vaultListDesc: "列出 vault 环境变量(值掩码显示)",
vaultRemoveDesc: "删除一个 vault 环境变量",
vaultKey: "变量名(字母、数字与下划线,不能以数字开头)",
vaultValue: "变量值,写入该 Agent 的 agent_state/.vault.toml",
},
run: { desc: "单次运行一个 Task", message: "本次 Task 的 Prompt" },
chat: {
desc: "打开交互式 REPL",
resume:
"恢复既有 Session 继续对话(缺省恢复当前 Agent 最近一次);Workspace 与模型沿用原 Session",
},
serve: {
serverDesc: "启动 Web 服务(HTTP API 与内置前端,同一进程)",
webDesc: "启动 Web 服务,就绪后用浏览器打开界面",
port: "监听端口(其次取环境变量 PORT,缺省 7364)",
host: "监听地址(其次取环境变量 HOST,缺省 127.0.0.1)",
noOpen: "不自动打开浏览器",
},
header,
chatHints: () =>
"输入消息发起对话;行尾 \\ 续行;/compact 压缩上下文;/exit 退出;Ctrl-C 中断对话。",
confirmExit: () => "确认退出 penguin?[y/N] ",
taskInterrupted: () => "[已中断当前对话]",
error: (message) => `[错误] ${message}`,
approvePrompt: () => "? 批准此工具调用?[Y/n] ",
taskStats: (s) =>
`[统计信息] 上下文 ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · 用时 ${s.elapsed} (${s.elapsedDelta})`,
abortLabel: (reason) => `[已中断]${reason ? `:${reason}` : ""}`,
reconnectLabel: (status, attempt) =>
`[重试] ${status === "timeout" ? "连接超时或网络中断" : "响应不完整或无法解析"},正在发起第 ${attempt} 次重试……`,
compactionStart: (mode, reason) =>
mode === "discard"
? `[压缩] 正在丢弃旧上下文(${reason})……`
: `[压缩] 正在总结压缩上下文(${reason})……`,
compactionStop: (mode, status, tokens) =>
(status === "completed"
? mode === "discard"
? "[压缩] 完成,旧上下文已丢弃"
: "[压缩] 完成,已切换到摘要后的新上下文"
: `[压缩] ${status === "aborted" ? "已中断" : "失败"},保留当前上下文`) +
(tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""),
compactNothing: () => "[压缩] 当前上下文为空,无需压缩",
approveModeInvalid: (value) =>
`无效的审批模式 "${value}"。请使用 allow-all、deny-all、read-only 或 always-ask。`,
approvalDecision: (decision) => (decision === "allow" ? "✓ [已批准]" : "× [已拒绝]"),
resumeNoOverride: () =>
"--resume 不接受 --workspace、--model-id 与 --provider:均沿用原 Session,创建后不可更换。",
resumeNoSession: () => "没有可恢复的 Session:当前 Agent 还没有任何会话记录。",
resumedBanner: (sessionId, messageCount) =>
`[已恢复] ${sessionId} · 当前上下文共 ${messageCount} 条消息`,
resumeHint: (command) => `继续本次对话:${command}`,
langInvalid: (value) => `无效的语言 "${value}"。请使用 en 或 zh。`,
langSet: (lang, rcPath) => `语言已设为 ${lang};已将 PENGUIN_LANG 写入 ${rcPath}。`,
langRestartConfirm: () => "现在打开新 shell 使其生效?[y/N] ",
langRestart: () => "正在打开使用新语言的新 shell(输入 exit 可返回)……",
langRestartHint: (rcPath) => `请打开新终端,或执行:source ${rcPath}`,
modelAdded: (model, def) => `已添加模型 ${model}。当前默认模型:${def ?? "(未设置)"}`,
modelUpdated: (model, def) => `已更新模型 ${model}。当前默认模型:${def ?? "(未设置)"}`,
defaultModelSet: (model) => `默认模型已设为 ${model}。`,
visionModelSet: (model) => `视觉模型已设为 ${model}。`,
modelListTitle: () => "已配置的模型:",
modelListEmpty: () => "尚未配置任何模型。用 `penguin config model add` 添加。",
vaultSet: (key) => `已保存 vault 条目 ${key}。`,
vaultRemoved: (key) => `已删除 vault 条目 ${key}。`,
vaultKeyMissing: (key) => `vault 条目 ${key} 不存在。`,
vaultListTitle: () => "vault 环境变量(值已掩码):",
vaultListEmpty: () => "vault 为空。用 `penguin config vault set` 添加。",
webReady: (url) => `Web 界面已就绪:${url}`,
webTimeout: (url) => `服务尚未就绪,请稍后手动打开 ${url}。`,
};
/** Get the message set for a language. */
export function getMessages(language: Language): Messages {
return language === "zh" ? zh : en;
}
/** Resolve the language from the env var and return its message set (the default used when no explicit `t` is given). */
export function defaultMessages(): Messages {
return getMessages(resolveLanguage());
}
/** Mask an API key: keep only a few trailing characters; return `-` when unconfigured. */
export function maskApiKey(apiKey: string | undefined): string {
if (!apiKey) return "-";
// Mask the whole thing when ≤12 chars: `****last4` reveals too much of a short secret (same threshold as the server-side mask).
if (apiKey.length <= 12) return "***";
return `****${apiKey.slice(-4)}`;
}
+46
View File
@@ -0,0 +1,46 @@
/**
* PenguinHarness CLI entry point.
*
* Only responsible for parsing CLI input into SDK arguments and rendering the streaming
* OmniMessage returned by the SDK.
* Loads .env on startup (e.g. locally configured ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL).
*
* penguin config model add|default ...
* penguin chat ...
* penguin run --message ...
* penguin server|web ...
* Docs: packages/docs/content/cli.{zh,en}.md (site path /docs/cli).
*/
import "dotenv/config";
import { Command } from "commander";
import { VERSION } from "@prismshadow/penguin-core";
import { registerConfigCommand } from "./commands/config.js";
import { registerRunCommand } from "./commands/run.js";
import { registerChatCommand } from "./commands/chat.js";
import { registerServeCommands } from "./commands/serve.js";
import { defaultMessages } from "./i18n.js";
// Language comes from the PENGUIN_LANG env var (default en); used consistently for
// command/option descriptions and runtime output.
const t = defaultMessages();
const program = new Command();
program
.name("penguin")
.description(t.cliDescription)
.version(VERSION, "-v, --version", t.versionDesc);
registerConfigCommand(program, t);
registerRunCommand(program, t);
registerChatCommand(program, t);
registerServeCommands(program, t);
// Show help only when no subcommand is given (empty input); do not error.
program.action(() => {
program.outputHelp();
});
program.parseAsync(process.argv).catch((err: unknown) => {
process.stderr.write(`${err instanceof Error ? err.message : String(err)}\n`);
process.exitCode = 1;
});
+121
View File
@@ -0,0 +1,121 @@
/**
* CLI input-layer helpers: multi-line input and paste support.
*
* - `PasteFilter`: a Transform inserted between stdin and readline. Once terminal bracketed
* paste mode is enabled, pasted content is wrapped in `\x1b[200~` … `\x1b[201~`; this
* Transform strips that pair of markers, withholds the pasted content in between (not
* forwarded to readline, so internal newlines aren't split into multiple submissions), and
* emits it as a whole via a `paste` event. All other keystrokes are forwarded to readline
* unchanged, preserving line editing and Ctrl-C.
* - `LineComposer`: assembles "line-by-line input + paste blocks" into one complete message.
* A single trailing backslash `\` means line continuation; a paste block goes into the
* pending buffer as a whole and is sent on Enter.
*/
import { Transform, type TransformCallback } from "node:stream";
const PASTE_START = "\x1b[200~";
const PASTE_END = "\x1b[201~";
/**
* Return the trailing part of `data` that could be a prefix of `marker` (hold, kept for
* concatenation with the next chunk); the rest is ready to process immediately (emit). Handles
* the case where a marker straddles a data-chunk boundary.
*/
export function splitTrailingPartial(data: string, marker: string): { emit: string; hold: string } {
const max = Math.min(marker.length - 1, data.length);
for (let k = max; k > 0; k--) {
if (data.endsWith(marker.slice(0, k))) {
return { emit: data.slice(0, data.length - k), hold: data.slice(data.length - k) };
}
}
return { emit: data, hold: "" };
}
export class PasteFilter extends Transform {
private inPaste = false;
private pasteBuf = "";
private leftover = "";
override _transform(chunk: Buffer | string, _enc: BufferEncoding, cb: TransformCallback): void {
let data = this.leftover + chunk.toString("utf8");
this.leftover = "";
while (data.length > 0) {
if (!this.inPaste) {
const i = data.indexOf(PASTE_START);
if (i === -1) {
const { emit, hold } = splitTrailingPartial(data, PASTE_START);
if (emit) this.push(emit);
this.leftover = hold;
data = "";
} else {
if (i > 0) this.push(data.slice(0, i));
data = data.slice(i + PASTE_START.length);
this.inPaste = true;
this.pasteBuf = "";
}
} else {
const j = data.indexOf(PASTE_END);
if (j === -1) {
const { emit, hold } = splitTrailingPartial(data, PASTE_END);
this.pasteBuf += emit;
this.leftover = hold;
data = "";
} else {
this.pasteBuf += data.slice(0, j);
data = data.slice(j + PASTE_END.length);
this.inPaste = false;
const text = this.pasteBuf;
this.pasteBuf = "";
this.emit("paste", text);
}
}
}
cb();
}
}
/** Whether the line ends in a continuation (an odd number of trailing backslashes; an even count is treated as escaped literal backslashes). */
export function endsWithContinuation(line: string): boolean {
const trailing = line.match(/(\\+)$/)?.[1] ?? "";
return trailing.length % 2 === 1;
}
/**
* Assembles line-by-line input and paste blocks into a complete message.
* `pushTypedLine` returns `{ message }` when a message is ready, or `{}` while still
* continuing/pending.
*/
export class LineComposer {
private pending: string[] = [];
pushTypedLine(line: string): { message?: string } {
if (endsWithContinuation(line)) {
this.pending.push(line.slice(0, -1));
return {};
}
if (this.pending.length > 0) {
const lines = line === "" ? this.pending : [...this.pending, line];
this.pending = [];
return { message: lines.join("\n") };
}
return { message: line };
}
/** Accept a paste block (strip trailing blank lines, normalize newlines); it goes into the pending buffer as a whole, waiting to be sent on Enter. */
pushPaste(text: string): { lineCount: number; normalized: string } {
const norm = text.replace(/\r\n?/g, "\n").replace(/\n+$/, "");
if (norm.length === 0) return { lineCount: 0, normalized: "" };
const lines = norm.split("\n");
this.pending.push(...lines);
return { lineCount: lines.length, normalized: norm };
}
hasPending(): boolean {
return this.pending.length > 0;
}
reset(): void {
this.pending = [];
}
}
+102
View File
@@ -0,0 +1,102 @@
/**
* Language persistence: write `PENGUIN_LANG` into the user's shell startup file, then restart
* the shell so it takes effect.
*
* A child process can't modify its parent shell's environment variables directly, so
* `penguin config lang` uses a "write the startup file + restart the shell" approach: write
* `export PENGUIN_LANG=<lang>` into the shell startup file inside a marked block (idempotent,
* updates in place), then open an interactive shell carrying the new language env var. New
* terminals will read the variable from the startup file, so it persists.
*/
import { spawn } from "node:child_process";
import { mkdir, readFile, writeFile } from "node:fs/promises";
import { dirname, join } from "node:path";
import type { Language } from "./i18n.js";
const BEGIN = "# >>> PenguinHarness PENGUIN_LANG >>>";
const END = "# <<< PenguinHarness PENGUIN_LANG <<<";
export type ShellKind = "zsh" | "bash" | "fish" | "unknown";
export interface ShellRc {
kind: ShellKind;
/** Absolute path to the startup file. */
rcPath: string;
/** Generate the export line for a given language (shell-syntax specific). */
body(lang: Language): string;
}
/** Resolve the startup file and export syntax from `$SHELL`. Falls back to `~/.profile` for an unknown shell. */
export function resolveShellRc(shell: string | undefined, home: string): ShellRc {
const base = (shell ?? "").split("/").pop()?.toLowerCase() ?? "";
if (base.includes("fish")) {
return {
kind: "fish",
rcPath: join(home, ".config", "fish", "config.fish"),
body: (lang) => `set -gx PENGUIN_LANG ${lang}`,
};
}
if (base.includes("zsh")) {
return {
kind: "zsh",
rcPath: join(home, ".zshrc"),
body: (lang) => `export PENGUIN_LANG=${lang}`,
};
}
if (base.includes("bash")) {
return {
kind: "bash",
rcPath: join(home, ".bashrc"),
body: (lang) => `export PENGUIN_LANG=${lang}`,
};
}
return {
kind: "unknown",
rcPath: join(home, ".profile"),
body: (lang) => `export PENGUIN_LANG=${lang}`,
};
}
/** Insert or update the marked PenguinHarness block in place within the text; leaves the rest of the content unchanged. */
export function upsertBlock(content: string, bodyLine: string): string {
const block = `${BEGIN}\n${bodyLine}\n${END}`;
const begin = content.indexOf(BEGIN);
const end = content.indexOf(END);
if (begin !== -1 && end !== -1 && end > begin) {
const before = content.slice(0, begin);
const after = content.slice(end + END.length);
return `${before}${block}${after}`;
}
// Append at the end: leave a blank line before it if there's existing content.
if (content.length === 0) return `${block}\n`;
const sep = content.endsWith("\n") ? "" : "\n";
return `${content}${sep}\n${block}\n`;
}
/** Write the language into the shell startup file (creating the directory if needed). Returns the file path written and the shell kind. */
export async function applyLanguageToRc(
lang: Language,
opts: { shell: string | undefined; home: string },
): Promise<{ rcPath: string; kind: ShellKind }> {
const rc = resolveShellRc(opts.shell, opts.home);
await mkdir(dirname(rc.rcPath), { recursive: true });
let content = "";
try {
content = await readFile(rc.rcPath, "utf8");
} catch {
/* File doesn't exist yet; treat as empty content */
}
await writeFile(rc.rcPath, upsertBlock(content, rc.body(lang)), "utf8");
return { rcPath: rc.rcPath, kind: rc.kind };
}
/** Open an interactive shell carrying the new language env var; this process exits when the user exits that shell. */
export function restartShell(lang: Language): void {
const shell = process.env.SHELL || "/bin/zsh";
const child = spawn(shell, ["-i"], {
stdio: "inherit",
env: { ...process.env, PENGUIN_LANG: lang },
});
child.on("exit", (code) => process.exit(code ?? 0));
child.on("error", () => process.exit(1));
}
+874
View File
@@ -0,0 +1,874 @@
/**
* CLI streaming renderer.
*
* Rendering rule: **only the streaming `partial_*` variants of model_msg are rendered**;
* complete (non-streaming) model_msg is never rendered. A complete message's content has
* already been delivered by its corresponding `partial_*` stream, so re-rendering it would
* be redundant. `partial_*` is written out token by token as it arrives.
* event_msg is not message rendering and is handled separately: `token_usage` accumulates
* and is summarized in the `[stats]` line at task end, `approval_decision` prints one line
* with the approval result, `abort` prints one line noting the interruption, and each of
* `compaction_begin`/`compaction_end` prints one line of compaction progress;
* `session_meta` is never rendered.
*
* **Screen lock (concurrent tools)**: tools run concurrently and asynchronously, so
* messages may arrive interleaved. The renderer queues internally to guarantee:
* - a streaming segment (the LLM's text/thinking/tool_call stream, or a given tool's
* output stream start->delta->stop) holds the screen until stop, while other messages
* queue up;
* - all output is locked while waiting for user input (the approval prompt,
* `beginUserPrompt`/`endUserPrompt`);
* - when the head of the queue is held, the holder's own subsequent messages are let
* through first (preserving in-segment order), avoiding deadlock.
*
* **Pairing tags**: a tool call and its output may be separated by several segments, so
* both are tagged with a shared word for pairing: the call line reads
* `[tool-653] $ cmd`, the output line `[tool-653] >> ...` (653 being the last 3
* characters of tool_call_id); nested (subagent) tools use
* `[agent-f2a-tool-653] $ cmd` (f2a being the last 3 characters of the direct child
* Session id). Approval lines carry no tag (they immediately follow the matching call
* line, so context makes the pairing clear): `[approved]`.
*
* **Nested sub-session messages** (those carrying an origin) are handled separately:
* child tool calls (so the user can see what the subagent is calling before approval)
* and child approval results are rendered, and child token_usage counts toward this
* task's delta and the Session total; everything else (child text/thinking, etc.) is
* not rendered — the child Agent's final text is already streamed through the parent
* tool's output gutter.
*
* No third-party color library is used; only minimal ANSI escapes.
*/
import { isEventMessage, isModelMessage } from "@prismshadow/penguin-core";
import type {
AbortPayload,
ApprovalDecision,
ApprovalDecisionPayload,
CompactionBeginPayload,
CompactionEndPayload,
MessageOrigin,
OmniMessage,
PartialTextPayload,
PartialThinkingPayload,
PartialToolCallPayload,
PartialToolCallOutputPayload,
RequestEndPayload,
TokenUsagePayload,
ToolCallPayload,
} from "@prismshadow/penguin-core";
import { renderPartialToolCall } from "./tool-render.js";
import { defaultMessages } from "./i18n.js";
import type { Messages } from "./i18n.js";
const DIM = "\x1b[2m";
const CYAN = "\x1b[36m";
const RESET = "\x1b[0m";
export function dim(text: string): string {
return `${DIM}${text}${RESET}`;
}
/** Colors a tool call line cyan, distinguishing it from body text/thinking (review comment #5). */
function cyan(text: string): string {
return `${CYAN}${text}${RESET}`;
}
/** Takes the last 3 characters of an id as the on-screen pairing number. */
function shortId(id: string): string {
return id.slice(-3);
}
/**
* On-screen pairing tag for a tool call/output: main-session tools ->
* `tool-<last 3 chars of id>`; nested (subagent) tools ->
* `agent-<last 3 chars of direct child Session>-tool-<last 3 chars of id>`.
*/
function callTag(toolCallId: string, origin?: readonly MessageOrigin[]): string {
const tid = `tool-${shortId(toolCallId)}`;
return origin && origin.length > 0 ? `agent-${shortId(origin[origin.length - 1]!)}-${tid}` : tid;
}
/** Converts a token count to a human-readable abbreviation: 1234->1.2k, 1500000->1.5M, <1000 unchanged. */
export function humanizeTokens(n: number): string {
const abs = Math.abs(n);
if (abs < 1000) return `${n}`;
if (abs < 1_000_000) {
const v = n / 1000;
return `${trimZero(v)}k`;
}
const v = n / 1_000_000;
return `${trimZero(v)}M`;
}
/** Keeps one decimal place but drops a trailing `.0`. */
function trimZero(v: number): string {
const s = v.toFixed(1);
return s.endsWith(".0") ? s.slice(0, -2) : s;
}
/** Adds an explicit sign to a delta string: non-negative gets a `+` prefix, negative already has its own `-` (context can go negative after compaction shrinks it). */
function signedDelta(formatted: string): string {
return formatted.startsWith("-") ? formatted : `+${formatted}`;
}
/** Converts milliseconds into a human-readable duration: `820ms`, `2.3s`, `1m3s`. */
function humanizeDuration(ms: number): string {
if (ms < 1000) return `${Math.round(ms)}ms`;
const s = ms / 1000;
if (s < 60) return `${trimZero(s)}s`;
const m = Math.floor(s / 60);
return `${m}m${Math.round(s % 60)}s`;
}
export function formatAbort(p: AbortPayload, t: Messages): string {
return dim(t.abortLabel(p.reason ?? undefined));
}
/**
* Statically renders resumed history messages (`--resume`: full-message semantics, no
* partial_*, including interrupted messages and their markers). Uses the
* same color scheme as streaming rendering: user input `> `, dim thinking, cyan tool
* calls, dim tool-output gutter; a message whose `stop_reason` isn't completed gets a
* dim marker appended at the end of its line.
*/
export function renderHistory(
messages: OmniMessage[],
out: NodeJS.WritableStream,
t: Messages = defaultMessages(),
): void {
for (const msg of messages) {
if (isEventMessage(msg)) {
const p = msg.payload as { type?: string } & AbortPayload;
if (p.type === "abort") out.write(`${formatAbort(p, t)}\n`);
continue;
}
if (!isModelMessage(msg)) continue;
const p = msg.payload as {
type?: string;
role?: string;
text?: string;
thinking?: string;
name?: string;
arguments?: string;
output?: string;
images?: string[];
tool_call_id?: string;
stop_reason?: string;
};
const marker = p.stop_reason && p.stop_reason !== "completed" ? dim(` [${p.stop_reason}]`) : "";
switch (p.type) {
case "text":
if (p.role === "user") out.write(`\n> ${p.text ?? ""}\n`);
else out.write(`${p.text ?? ""}${marker}\n`);
break;
case "image_url":
out.write(`\n> ${dim("[image]")}\n`);
break;
case "thinking":
out.write(`${dim(p.thinking ?? "")}${marker}\n`);
break;
case "tool_call": {
const preview =
renderPartialToolCall(p.name ?? "", p.arguments ?? "") ?? `${p.name} ${p.arguments}`;
out.write(`${cyan(`[${callTag(p.tool_call_id ?? "")}] ${preview}`)}${marker}\n`);
break;
}
case "tool_call_output": {
const tag = callTag(p.tool_call_id ?? "");
for (const line of (p.output ?? "").split("\n")) {
out.write(`${DIM}[${tag}] >> ${RESET}${line}\n`);
}
// Attached images aren't rendered by the terminal; print one placeholder line per image.
for (const _ of p.images ?? []) {
out.write(`${DIM}[${tag}] >> [image]${RESET}\n`);
}
break;
}
default:
break; // inline_data / inline_thinking etc.: not shown in static history rendering for now
}
}
}
/**
* Streaming renderer: writes the OmniMessage stream to the output stream. The display
* text for tool calls is decided locally by `tool-render.ts`; it no longer accepts a
* tool-render callback from core (rendering has moved down into the CLI).
*/
export class StreamRenderer {
private readonly out: NodeJS.WritableStream;
private readonly t: Messages;
/** Pending render queue: while the screen is held (a streaming segment is in progress / awaiting user input), messages queue up here. */
private pending: OmniMessage[] = [];
/** The streaming segment currently holding the screen ("llm" or "out:<tool_call_id>"); null = idle. */
private holder: string | null = null;
/** Awaiting user input (approval prompt): locks the screen, all messages queue up. */
private promptActive = false;
/** Key of the call the current interactive prompt belongs to (the tool_call passed to beginUserPrompt); null = unattached. */
private promptKey: string | null = null;
/**
* Approval results for **other calls** that arrive during an interactive prompt
* (concurrent subagent / auto-approval paths): must not be written straight into the
* middle of an unanswered prompt, so they're deferred and rendered in order once
* endUserPrompt unlocks the screen.
*/
private deferredDecisions: Array<{
toolCall: OmniMessage<ToolCallPayload>;
decision: ApprovalDecision;
}> = [];
/** Reentrancy guard for drain. */
private draining = false;
/**
* Keys (origin chain + tool_call_id) of call lines already **rendered in place** from
* a complete message: rendered ahead of the streaming copy at approval time, so any
* streaming/nested copy that arrives afterward is deduplicated and skipped based on
* this set. Guarantees the approval prompt always immediately follows its matching
* call line (messages arrive through an async pipeline and may arrive later than the
* approval callback). Cleared at task end (see endTask).
*/
private ensuredCallLines = new Set<string>();
/** Call-line key of the last **content line actually written**; cleared once anything else is written. Used to check whether a call line is still adjacent to the current position. */
private lastLineKey: string | null = null;
/** Calls whose result has already been rendered in place at the approval callback (keyed the same as callLineKey); deduplicates a later-arriving approval_decision event. */
private renderedDecisions = new Set<string>();
/** Whether we're currently mid-way through a streaming line (text/thinking/tool output) that hasn't been newline-terminated yet. */
private inLine = false;
/** Whether we're currently in a dim span (thinking), used to know when to emit RESET. */
private inDim = false;
/** Whether tool-call output is at the start of a line (decides whether the gutter needs to be written). */
private toolOutLineStart = true;
/** Buffer for partial_tool_call; each delta streams out the newly appended suffix of the preview. */
private partialToolCalls = new Map<
string,
{ name: string; arguments: string; lastPreview: string }
>();
/** The partial_tool_call currently being rendered as a stream. */
private partialToolCallLineId: string | null = null;
/** This task's accumulated request tokens, the parent session's cumulative Session tokens, and whether this task has seen any usage. */
private taskTokens = 0;
private sessionTotal = 0;
private hasUsage = false;
/**
* Session-level accumulation of sub-session (subagent) request tokens: persists across
* tasks, never reset by endTask. The Token total shown to the user =
* sessionTotal + subagentTotal, using the same accounting as this task's delta
* (parent + child), guaranteeing the sum of per-task deltas never exceeds the
* cumulative increase.
*/
private subagentTotal = 0;
/** Current context (= input+output = total of the most recent request), the context at the end of the previous task, and cumulative Session elapsed time (ms). */
private contextNow = 0;
private contextAtTaskStart = 0;
private sessionElapsedMs = 0;
/**
* Compaction in progress (between a pair of parent-session compaction events): any
* parent-session token_usage arriving during this window is compaction-request usage —
* it does not update the context accounting (the actual usage after compaction is
* reported by the next normal request); it's accumulated into compactionTokens so the
* compaction-completion line can show "usage this time", and also staged into
* pendingCompactionTokens pending final attribution (see below).
*/
private compactionActive = false;
private compactionTokens = 0;
/**
* Staged compaction usage: when a compaction event arrives, it's not yet known whether
* it happened **mid-turn** (a normal request_end still follows in this turn ->
* attribute to this turn) or **after the turn ended** (nothing follows -> don't
* attribute to this turn). Mid-turn compaction is folded into taskTokens at the next
* non-compaction request_end; compaction after the turn ended is discarded when
* endTask/endCompact settles up. Uses the same accounting as the Web side
* (stream-model / task-stats).
*/
private pendingCompactionTokens = 0;
/**
* Timestamps (ms) of this task's first (non-session_meta) message and its last
* **non-compaction** request_end: the elapsed time shown in the stats line = the
* latter minus the former. A mid-turn compaction naturally falls within this span and
* is counted; one after the turn ends falls after it and is naturally excluded
* (consistent with "the last request_end before stats were queried"). The degenerate
* case of a turn with no request_end at all falls back to the externally supplied
* wall-clock elapsed time.
*/
private taskFirstTsMs: number | null = null;
private taskLastReqEndMs: number | null = null;
/** Terminal state (timeout/malformed) of the previous request: the next request_begin is a retry, at which point a notice is printed. */
private pendingRetry: "timeout" | "malformed" | null = null;
/** Number of retries already initiated (increments on consecutive failures, reset once a request completes normally). */
private reconnectRun = 0;
constructor(out: NodeJS.WritableStream = process.stdout, t: Messages = defaultMessages()) {
this.out = out;
this.t = t;
}
handle(msg: OmniMessage): void {
this.pending.push(msg);
this.drain();
}
/**
* Enters user interaction (approval prompt): first ensures the call line awaiting
* approval is **immediately adjacent to the current position** (if unrendered or
* separated by other output, render it in place from the complete message directly),
* then finishes the current line and locks the screen, queuing any messages that
* arrive in the meantime — guaranteeing "tool call -> approval prompt" stay adjacent,
* for both the main Agent and subagents.
*/
beginUserPrompt(toolCall?: OmniMessage<ToolCallPayload>): void {
if (toolCall) this.ensureAdjacentCallLine(toolCall);
this.finishLine();
this.promptActive = true;
this.promptKey = toolCall
? this.callLineKey(toolCall.payload.tool_call_id, toolCall.origin)
: null;
}
/**
* Renders one approval result, guaranteeing "tool call -> (approval prompt) ->
* approval result" appear consecutively:
* - interactive path: called **before** the prompt ends and unlocks (nothing else can
* preempt output while the lock is held);
* - auto-approval path (allow-all etc., no prompt): if the call line isn't adjacent,
* render it in place first, then write the result, so they appear as a pair.
* Idempotent (a given call's result is rendered only once); a subsequent
* approval_decision event arriving through the pipeline is deduplicated by key.
*/
noteApprovalDecision(toolCall: OmniMessage<ToolCallPayload>, decision: ApprovalDecision): void {
const key = this.callLineKey(toolCall.payload.tool_call_id, toolCall.origin);
// The screen is locked by **another call's** interactive prompt (e.g. auto-approval
// of a concurrent subagent): must not write straight into the middle of an
// unanswered prompt, so defer until unlocked; this prompt's own result still renders
// in place as usual (it holds the lock).
if (this.promptActive && this.promptKey !== key) {
this.deferredDecisions.push({ toolCall, decision });
return;
}
if (this.renderedDecisions.has(key)) return;
this.renderedDecisions.add(key);
this.ensureAdjacentCallLine(toolCall);
this.finishLine();
this.out.write(`${dim(this.t.approvalDecision(decision))}\n`);
this.lastLineKey = null;
}
/** Call-line dedup key: origin chain + tool_call_id (parent/child session ids may collide, so the chain is needed to disambiguate). */
private callLineKey(id: string, origin?: readonly MessageOrigin[]): string {
return `${origin?.join("/") ?? ""}:${id}`;
}
/**
* Ensures a given tool_call's call line is adjacent to the current position: if it
* isn't the last content line (unrendered, or separated by other output since), it is
* (re-)rendered in place from the complete message, and registered so any late
* streaming/nested copy is deduplicated and skipped.
*/
private ensureAdjacentCallLine(tc: OmniMessage<ToolCallPayload>): void {
const key = this.callLineKey(tc.payload.tool_call_id, tc.origin);
// The call line is already the last content line and its streaming segment has
// already finished: already adjacent, nothing to do. If it's still mid-stream (the
// line may show only half the arguments), re-render the full line in place and
// register it for dedup — otherwise a late tail delta arriving after unlock would
// start a duplicate call line, breaking the "call -> prompt -> result" adjacency
// invariant.
if (this.lastLineKey === key && this.partialToolCallLineId !== tc.payload.tool_call_id) {
return;
}
this.renderCallLine(tc.payload, tc.origin, key);
}
/** Renders one call line in place from a complete tool_call and registers its dedup key (shared by in-place approval rendering and nested rendering). */
private renderCallLine(
p: ToolCallPayload,
origin: readonly MessageOrigin[] | undefined,
key: string,
): void {
this.ensuredCallLines.add(key);
const preview = renderPartialToolCall(p.name, p.arguments) ?? `${p.name} ${p.arguments}`;
this.finishLine();
this.out.write(`${cyan(`[${callTag(p.tool_call_id, origin)}] ${preview}`)}\n`);
this.lastLineKey = key;
}
/** User interaction ends: unlocks the screen, first renders approval results deferred during the lock, then drains the queue. */
endUserPrompt(): void {
this.promptActive = false;
this.promptKey = null;
this.flushDeferredDecisions();
this.drain();
}
/** Renders approval results deferred during the interactive prompt (call line + result as a pair; called after unlocking). */
private flushDeferredDecisions(): void {
const deferred = this.deferredDecisions;
if (deferred.length === 0) return;
this.deferredDecisions = [];
for (const d of deferred) this.noteApprovalDecision(d.toolCall, d.decision);
}
/** Streaming segment ownership: the LLM stream (text/thinking/tool_call share one stream serially) or a given tool's output stream; null = atomic message. */
private streamOwner(msg: OmniMessage): string | null {
if (msg.origin && msg.origin.length > 0) return null; // nested messages render as atomic lines
if (!isModelMessage(msg)) return null;
const type = msg.payload.type;
if (type === "partial_text" || type === "partial_thinking" || type === "partial_tool_call") {
return "llm";
}
if (type === "partial_tool_call_output") {
return `out:${(msg.payload as PartialToolCallOutputPayload).tool_call_id}`;
}
return null;
}
private isStop(msg: OmniMessage): boolean {
return (msg.payload as { event_type?: string }).event_type === "stop";
}
/**
* Drains the pending render queue. The same streaming segment (start->delta->stop)
* holds the screen until stop, while other messages queue up; while the screen is
* held, the holder's own subsequent messages are let through first (preserving
* in-segment order, while other messages keep their arrival order); nothing is let
* through while awaiting user input.
*/
private drain(): void {
if (this.draining) return;
this.draining = true;
try {
while (!this.promptActive && this.pending.length > 0) {
if (this.holder === null) {
const msg = this.pending.shift()!;
const owner = this.streamOwner(msg);
if (owner !== null) this.holder = this.isStop(msg) ? null : owner;
this.renderNow(msg);
continue;
}
// Screen is held: let through all of the holder's own messages in a single
// pass (avoiding the quadratic cost of rescanning from the queue head after
// each message); once the holder releases mid-scan (stop), put the remaining
// messages back in original order, returning to plain FIFO.
const keep: OmniMessage[] = [];
let progressed = false;
for (let i = 0; i < this.pending.length; i++) {
if (this.promptActive || this.holder === null) {
keep.push(...this.pending.slice(i));
break;
}
const msg = this.pending[i]!;
if (this.streamOwner(msg) === this.holder) {
if (this.isStop(msg)) this.holder = null;
this.renderNow(msg);
progressed = true;
} else {
keep.push(msg);
}
}
this.pending = keep;
if (!progressed) break; // no message from the holder in the queue: wait for it to arrive
}
} finally {
this.draining = false;
}
}
/** Actually renders one message (queue scheduling is already done by drain). */
private renderNow(msg: OmniMessage): void {
if (msg.origin && msg.origin.length > 0) {
this.handleNested(msg);
return;
}
// The timestamp of this task's first (non-session_meta) message = the start point for
// the stats-line elapsed time. session_meta can predate this turn by a long time (a
// session may sit idle for a day before the first question), so it is excluded,
// matching Web / Trace accounting.
if (this.taskFirstTsMs === null && msg.type !== "session_meta") {
const ms = Date.parse(msg.timestamp);
if (Number.isFinite(ms)) this.taskFirstTsMs = ms;
}
if (isModelMessage(msg)) {
const payload = msg.payload;
switch (payload.type) {
case "partial_text":
this.handlePartialText(payload as PartialTextPayload);
return;
case "partial_thinking":
this.handlePartialThinking(payload as PartialThinkingPayload);
return;
case "partial_tool_call":
this.handlePartialToolCall(payload as PartialToolCallPayload);
return;
case "partial_tool_call_output":
this.handlePartialToolOutput(payload as PartialToolCallOutputPayload);
return;
// Complete (non-streaming) model_msg is never rendered (including image_url/inline_*); the content has already been shown by partial_*.
default:
return;
}
}
if (isEventMessage(msg)) {
const payload = msg.payload;
if (payload.type === "token_usage") {
// Accumulate this task's usage, printed together when the task ends (endTask),
// not shown after every tool call/round.
const p = payload as TokenUsagePayload;
this.sessionTotal = p.session.total;
if (this.compactionActive) {
// Usage of a compaction request: staged first (final attribution depends on
// whether a normal request_end still follows in this turn), and accumulated
// into compactionTokens so the compaction-completion line can show "usage this
// time"; does not update context accounting (see the compactionActive comment).
this.pendingCompactionTokens += p.request.total;
this.compactionTokens += p.request.total;
} else {
this.taskTokens += p.request.total;
this.contextNow = p.request.total; // current context = total of the most recent normal request
this.hasUsage = true;
}
} else if (payload.type === "approval_decision") {
// The approval result has usually already been rendered in place at the
// approval callback (noteApprovalDecision, guaranteeing three consecutive
// lines); deduplicated here by key; falls back to rendering one line (without a
// pairing tag) if it wasn't rendered yet.
const p = payload as ApprovalDecisionPayload;
if (this.renderedDecisions.delete(this.callLineKey(p.tool_call_id))) return;
this.finishLine();
this.out.write(`${dim(this.t.approvalDecision(p.decision))}\n`);
this.lastLineKey = null;
} else if (payload.type === "abort") {
// Run ended (user interrupt / retries exhausted): clear any pending retry state so the next run doesn't mistakenly print a retry line.
this.pendingRetry = null;
this.reconnectRun = 0;
this.finishLine();
this.out.write(`${formatAbort(payload as AbortPayload, this.t)}\n`);
this.lastLineKey = null;
} else if (payload.type === "request_begin") {
// The previous request ended in timeout/malformed -> this request is a retry
// carrying <turn_retried>: printed when the retry **actually starts** (when
// retries are exhausted, there's no retry after the last failure, only an abort
// explaining why).
if (this.pendingRetry) {
this.reconnectRun += 1;
this.finishLine();
this.out.write(`${dim(this.t.reconnectLabel(this.pendingRetry, this.reconnectRun))}\n`);
this.lastLineKey = null;
this.pendingRetry = null;
}
} else if (payload.type === "request_end") {
const p = payload as RequestEndPayload;
if (!this.compactionActive) {
// A non-compaction request_end = the end of the turn so far: records the
// timestamp (the end point for elapsed time), and settles any previously
// staged compaction usage — reaching here means that compaction was followed
// by a normal Request in this turn (mid-turn compaction), so its usage is
// attributed to this turn.
const ms = Date.parse(msg.timestamp);
if (Number.isFinite(ms)) this.taskLastReqEndMs = ms;
if (this.pendingCompactionTokens > 0) {
this.taskTokens += this.pendingCompactionTokens;
this.pendingCompactionTokens = 0;
this.hasUsage = true;
}
}
if (p.status === "timeout" || p.status === "malformed") {
this.pendingRetry = p.status;
} else {
this.pendingRetry = null;
this.reconnectRun = 0;
}
} else if (payload.type === "compaction_begin") {
// Paired compaction events: begin signals compaction is in progress.
const p = payload as CompactionBeginPayload;
this.finishLine();
this.compactionActive = true;
this.compactionTokens = 0;
this.out.write(`${dim(this.t.compactionStart(p.mode, p.reason))}\n`);
this.lastLineKey = null;
} else if (payload.type === "compaction_end") {
// end signals the result and shows the tokens consumed by the compaction request (if any).
const p = payload as CompactionEndPayload;
this.finishLine();
this.compactionActive = false;
// Same accounting as the stats line: total = Session cumulative (parent + child), delta = usage of this compaction.
const tokens =
this.compactionTokens > 0
? {
total: humanizeTokens(this.sessionTotal + this.subagentTotal),
delta: signedDelta(humanizeTokens(this.compactionTokens)),
}
: undefined;
this.compactionTokens = 0;
this.out.write(`${dim(this.t.compactionStop(p.mode, p.status, tokens))}\n`);
this.lastLineKey = null;
}
return;
}
// session_meta: not rendered.
}
/**
* Nested sub-session messages (carrying an origin): renders the child tool call
* (tagged `agent-xxx-tool-xxx` to mark it as coming from a subagent) and its approval
* result; the request delta of a child token_usage counts toward this task's usage;
* everything else is not rendered (see the rendering rule at the top of this file).
*/
private handleNested(msg: OmniMessage): void {
const origin = msg.origin!;
if (isModelMessage(msg)) {
if (msg.payload.type === "tool_call") {
// A complete tool_call renders one line (nested messages never render
// partial_*, so there's no duplication); one already rendered in place at
// approval time (message arrived later than the approval callback) is
// deduplicated by key and skipped.
const p = msg.payload as ToolCallPayload;
const key = this.callLineKey(p.tool_call_id, origin);
if (this.ensuredCallLines.has(key)) return;
this.renderCallLine(p, origin, key);
}
return;
}
if (isEventMessage(msg)) {
if (msg.payload.type === "approval_decision") {
// The approval result is usually already rendered in place at the approval callback; deduplicated here by key; falls back to rendering if it wasn't rendered yet.
const p = msg.payload as ApprovalDecisionPayload;
if (this.renderedDecisions.delete(this.callLineKey(p.tool_call_id, origin))) {
return;
}
this.finishLine();
this.out.write(`${dim(this.t.approvalDecision(p.decision))}\n`);
this.lastLineKey = null;
} else if (msg.payload.type === "token_usage") {
// Child-session usage counts toward this task's Token delta and the Session total (parent and child use the same accounting); context still follows parent-session accounting.
const req = (msg.payload as TokenUsagePayload).request.total;
this.taskTokens += req;
this.subagentTotal += req;
this.hasUsage = true;
}
}
}
private handlePartialText(p: PartialTextPayload): void {
if (p.event_type === "stop") {
this.finishLine();
return;
}
// Insert a line break when switching from thinking (dim) to body text, to avoid them running together.
if (this.inDim) this.finishLine();
if (p.text) {
this.out.write(p.text);
this.inLine = true;
this.lastLineKey = null;
}
}
private handlePartialThinking(p: PartialThinkingPayload): void {
if (p.event_type === "stop") {
this.finishLine();
return;
}
if (!this.inDim) {
this.out.write(DIM);
this.inDim = true;
}
if (p.thinking) {
this.out.write(p.thinking);
this.inLine = true;
this.lastLineKey = null;
}
}
private handlePartialToolCall(p: PartialToolCallPayload): void {
// The call line was already rendered in place from the complete message at approval time: skip the whole late-arriving streaming copy (clean up the buffer on stop).
if (this.ensuredCallLines.has(this.callLineKey(p.tool_call_id))) {
if (p.event_type === "stop") this.partialToolCalls.delete(p.tool_call_id);
return;
}
let partial = this.partialToolCalls.get(p.tool_call_id);
if (!partial) {
if (p.event_type === "stop") return;
partial = { name: p.name, arguments: "", lastPreview: "" };
this.partialToolCalls.set(p.tool_call_id, partial);
}
if (p.name) partial.name = p.name;
if (p.arguments) {
partial.arguments += p.arguments;
}
if (p.event_type === "stop") {
if (partial.lastPreview) this.finishLine();
this.partialToolCalls.delete(p.tool_call_id);
return;
}
if (!p.arguments) return;
if (this.inDim) this.finishLine();
const preview = renderPartialToolCall(partial.name, partial.arguments);
if (preview === null) return;
// The line starts with a pairing tag [tool-<last 3 chars of id>], matching the output line that follows.
const key = this.callLineKey(p.tool_call_id);
if (this.partialToolCallLineId !== p.tool_call_id) {
this.finishLine();
this.partialToolCallLineId = p.tool_call_id;
this.out.write(cyan(`[${callTag(p.tool_call_id)}] ${preview}`));
} else if (preview.startsWith(partial.lastPreview)) {
this.out.write(cyan(preview.slice(partial.lastPreview.length)));
} else {
// The preview usually grows monotonically with the arguments; if escaping/folding makes it non-appendable, start a new line with the current readable state.
this.finishLine();
this.partialToolCallLineId = p.tool_call_id;
this.out.write(cyan(`[${callTag(p.tool_call_id)}] ${preview}`));
}
partial.lastPreview = preview;
this.inLine = true;
this.lastLineKey = key;
}
private handlePartialToolOutput(p: PartialToolCallOutputPayload): void {
if (p.event_type === "stop") {
this.finishLine();
return;
}
if (this.inDim) this.finishLine();
if (p.output) this.writeToolOutput(p.output, callTag(p.tool_call_id));
// Image delta (carried whole in a single delta): the terminal doesn't render the
// image itself, so print one placeholder line per image, using the same pairing tag
// as the output gutter.
if (p.images && p.images.length > 0) {
this.finishLine();
const tag = callTag(p.tool_call_id);
for (const _ of p.images) {
this.out.write(`${DIM}[${tag}] >> [image]${RESET}\n`);
}
this.lastLineKey = null;
}
}
/**
* Writes tool-call **output** line by line, each line starting with the dim gutter
* `[tool-<last 3 chars of id>] >> `, paired with the call line (cyan `[tool-xxx] $
* cmd`). Streaming chunks arrive incrementally; whether to write the gutter is
* decided by the current line-start state.
*/
private writeToolOutput(chunk: string, tag: string): void {
let i = 0;
while (i < chunk.length) {
if (this.toolOutLineStart) {
this.out.write(`${DIM}[${tag}] >> ${RESET}`);
this.toolOutLineStart = false;
this.inLine = true;
}
const nl = chunk.indexOf("\n", i);
if (nl === -1) {
this.out.write(chunk.slice(i));
i = chunk.length;
} else {
this.out.write(chunk.slice(i, nl + 1));
this.toolOutLineStart = true;
this.inLine = false;
i = nl + 1;
}
}
this.lastLineKey = null;
}
/**
* Task end: forcibly releases the screen lock and drains any remaining messages
* (normally every streaming segment has already closed), finishes the current line,
* and prints one line of stats — all as Session cumulative values + this task's
* delta: context (input+output of the most recent request; delta = minus the context
* at the start of this task, which can be negative once compaction shrinks context),
* Token (Session cumulative = parent-session cumulative + child-session cumulative;
* delta = added this task, same accounting for parent and child), elapsed time
* (Session total elapsed; delta = this task's elapsed). This task's counters are then
* reset.
*/
endTask(elapsedMs = 0): void {
this.promptActive = false;
this.promptKey = null;
this.flushDeferredDecisions();
this.holder = null;
this.drain();
this.finishLine();
// This task's elapsed time = first message -> last non-compaction request_end
// (mid-turn compaction falls within the span and is counted; compaction after the
// turn ends falls after it and isn't). The degenerate case of a turn with no
// request_end at all (e.g. aborted before the first Request even ran) falls back to
// the externally supplied wall-clock elapsedMs. Any staged but unsettled compaction
// usage is discarded here (compaction after the turn ended isn't attributed to it).
const elapsed =
this.taskFirstTsMs !== null && this.taskLastReqEndMs !== null
? Math.max(0, this.taskLastReqEndMs - this.taskFirstTsMs)
: elapsedMs;
this.sessionElapsedMs += elapsed;
if (this.hasUsage) {
const contextDelta = this.contextNow - this.contextAtTaskStart;
this.out.write(
`${dim(
this.t.taskStats({
context: humanizeTokens(this.contextNow),
contextDelta: signedDelta(humanizeTokens(contextDelta)),
tokens: humanizeTokens(this.sessionTotal + this.subagentTotal),
tokensDelta: signedDelta(humanizeTokens(this.taskTokens)),
elapsed: humanizeDuration(this.sessionElapsedMs),
elapsedDelta: signedDelta(humanizeDuration(elapsed)),
}),
)}\n`,
);
this.contextAtTaskStart = this.contextNow;
this.lastLineKey = null;
}
this.taskTokens = 0;
this.pendingCompactionTokens = 0;
this.taskFirstTsMs = null;
this.taskLastReqEndMs = null;
this.hasUsage = false;
// Compaction always closes within run/compact (stop is always reached); this is a
// defensive reset to prevent state from leaking into the next task on an
// exceptional path.
this.compactionActive = false;
this.compactionTokens = 0;
// Dedup/buffer registrations are only meaningful within this task: clear them to prevent unbounded growth in long sessions (chat).
this.ensuredCallLines.clear();
this.renderedDecisions.clear();
this.partialToolCalls.clear();
}
/**
* Cleans up after a manual `/compact` (outside a Task boundary): compaction usage has
* already been shown on the compaction-completion line and counted into the Session
* total, so no stats line is printed here; only settles the Session elapsed time and
* resets this task's counters — otherwise the compaction's usage would remain in
* taskTokens and be mistakenly counted into the next task's `[stats]` delta (or never
* settled at all if the user exits right after).
*/
endCompact(elapsedMs = 0): void {
this.sessionElapsedMs += elapsedMs;
this.taskTokens = 0;
this.pendingCompactionTokens = 0;
this.taskFirstTsMs = null;
this.taskLastReqEndMs = null;
this.hasUsage = false;
this.compactionActive = false;
this.compactionTokens = 0;
}
private closeDim(): void {
if (this.inDim) {
this.out.write(RESET);
this.inDim = false;
}
}
/** Finishes the current streaming line: closes dim mode, emits a trailing newline, and resets tool output to line-start. */
private finishLine(): void {
this.closeDim();
if (this.inLine) {
this.out.write("\n");
this.inLine = false;
}
this.toolOutLineStart = true;
this.partialToolCallLineId = null;
}
}
+107
View File
@@ -0,0 +1,107 @@
/**
* Consumption loop that drives a Task to completion (CLI side, shared by run and chat).
*
* New protocol: `session.run(prompt, { signal, approve })` runs the entire ReAct loop in one
* call — within a turn, the engine invokes the `approve` callback for each tool_call, executing
* it on allow, with execution possibly overlapping. The CLI only needs to consume the output
* stream and supply `approve`. The approval strategy is determined by the permission mode
* (allow-all / deny-all / read-only / always-ask per-call approval).
*/
import { isEventMessage } from "@prismshadow/penguin-core";
import type { ApproveFn, OmniMessage, Session } from "@prismshadow/penguin-core";
import type { StreamRenderer } from "./render.js";
import { makeApprove, promptApproval, type ApprovalMode } from "./approval.js";
import type { Messages } from "./i18n.js";
export interface RunTaskOptions {
/** Approval mode (default allow-all). */
mode?: ApprovalMode;
/** Interrupt signal (Ctrl-C, etc.). */
signal?: AbortSignal;
renderer: StreamRenderer;
/** The actual Q&A for interactive approval; defaults to the one-off `promptApproval`. */
interactivePrompt?: ApproveFn;
/** Message set. */
t: Messages;
}
/** Result of one Task: `aborted` = the Task ended with an abort event (LLM failure/reconnect exhausted/user interrupt). */
export interface RunTaskResult {
aborted: boolean;
}
export async function runTask(
session: Session,
prompt: OmniMessage[],
opts: RunTaskOptions,
): Promise<RunTaskResult> {
const basePrompt: ApproveFn = opts.interactivePrompt ?? (() => promptApproval({ t: opts.t }));
// Lock the renderer while waiting for the user's approval input: messages from concurrent
// tools/subsessions are queued and released together once the Q&A finishes, so the prompt
// isn't scrambled by later output. The pending tool_call is passed in so its call line stays
// right before the prompt; the approval result is rendered in place **before unlocking** —
// "tool call → approval prompt → approval result" stays three consecutive lines, for both
// the main Agent and subagents (messages arriving via the async pipeline may lag behind the
// approval callback, hence render-in-place plus de-duplication of the copy).
//
// Serialization: the parent session and a run_subagent child session share this callback and
// may request approval concurrently (the parent is waiting on one approval while an
// already-approved child session starts its own). Concurrent prompts would clobber the same
// Q&A state and fight over the same stdin (one answer resolving two questions, leaving the
// other permanently stuck); a promise chain queues them so only one question is asked at a
// time.
let promptChain: Promise<unknown> = Promise.resolve();
const interactivePrompt: ApproveFn = (tc) => {
const result = promptChain.then(async () => {
opts.renderer.beginUserPrompt(tc);
try {
const decision = await basePrompt(tc);
opts.renderer.noteApprovalDecision(tc, decision);
return decision;
} finally {
opts.renderer.endUserPrompt();
}
});
promptChain = result.then(
() => undefined,
() => undefined,
);
return result;
};
const approveByMode = makeApprove({
mode: opts.mode ?? "allow-all",
toolPermission: (name) => session.toolPermission(name),
interactivePrompt,
});
// The auto-approval path (allow-all / deny-all / read-only approvals) has no prompt: it
// likewise renders the "call line → approval result" pair in place; the interactive path's
// already-rendered copy is idempotently de-duplicated inside note.
const approve: ApproveFn = async (tc) => {
const decision = await approveByMode(tc);
opts.renderer.noteApprovalDecision(tc, decision);
return decision;
};
// A single run drives the whole ReAct loop (the engine requests approval per call and runs
// tools concurrently within a turn). Once the task ends (including on error), endTask
// prints this task's stats (context/Token/elapsed time). The engine collapses failures
// (auth errors, reconnect exhausted, etc.) into a main-session abort event rather than
// throwing; the result reported here reflects that, for `penguin run` to map to
// an exit code.
const startedAt = Date.now();
let aborted = false;
try {
for await (const msg of session.run(prompt, {
approve,
...(opts.signal ? { signal: opts.signal } : {}),
})) {
if (isEventMessage(msg) && msg.payload.type === "abort" && (msg.origin?.length ?? 0) === 0) {
aborted = true;
}
opts.renderer.handle(msg);
}
} finally {
opts.renderer.endTask(Date.now() - startedAt);
}
return { aborted };
}
+151
View File
@@ -0,0 +1,151 @@
/**
* Streaming tool-call rendering (CLI side).
*
* The CLI only consumes `partial_tool_call` for visible rendering. exec_command is shown as
* `$ <cmd>` as early as possible; input_command / input_subagent show the target session id,
* with a non-empty payload (chars / prompt) appended as `<< <content>` — the payload is
* critical for approval and later audit (writing to stdin is equivalent to running a command),
* so the session id alone is not enough; run_subagent shows the prompt; other tools fall back
* to `name(args-prefix)`.
*
* The render layer streams by appending to the preview (see render.ts), so the preview format
* must stay append-only: rendering only starts once the target id has fully appeared, the
* payload is only appended at the end, and the preview stops growing once it hits the
* truncation limit.
*/
/** Max length of the single-line preview for a payload (chars / prompt); truncated with an ellipsis beyond this, after which the preview stops growing. */
const MAX_PAYLOAD_PREVIEW = 120;
/** Collapse to a single line: newlines/runs of whitespace become a single space, and leading/trailing whitespace is trimmed. */
function toSingleLine(text: string): string {
return text.replace(/\s+/g, " ").trim();
}
/**
* Turn control characters into a visible, faithful form so stdin writes don't garble the
* screen: `\n`/`\r`/`\t` are shown as escape literals (whether Enter was pressed is important
* information and must not collapse into a space), other C0 control chars and DEL use caret
* notation (U+0003 → `^C`); backslash itself is escaped to avoid ambiguity.
*/
function visualizeControlChars(text: string): string {
return text.replace(/[\\\u0000-\u001f\u007f]/g, (ch) => {
if (ch === "\\") return "\\\\";
if (ch === "\n") return "\\n";
if (ch === "\r") return "\\r";
if (ch === "\t") return "\\t";
if (ch === "\u007f") return "^?";
return `^${String.fromCharCode(ch.charCodeAt(0) + 64)}`;
});
}
/** Truncate to the single-line preview limit, appending an ellipsis if exceeded. */
function capPreview(text: string): string {
return text.length > MAX_PAYLOAD_PREVIEW ? `${text.slice(0, MAX_PAYLOAD_PREVIEW)}…` : text;
}
/** Extract the current value of a string field from a possibly-incomplete JSON object string. */
function extractPartialStringField(argsJson: string, field: string): string | null {
const key = `"${field}"`;
const keyIndex = argsJson.indexOf(key);
if (keyIndex === -1) return null;
let i = keyIndex + key.length;
while (/\s/.test(argsJson[i] ?? "")) i += 1;
if (argsJson[i] !== ":") return null;
i += 1;
while (/\s/.test(argsJson[i] ?? "")) i += 1;
if (argsJson[i] !== '"') return null;
i += 1;
let out = "";
let escaped = false;
for (; i < argsJson.length; i += 1) {
const ch = argsJson[i]!;
if (escaped) {
switch (ch) {
case "n":
out += "\n";
break;
case "r":
out += "\r";
break;
case "t":
out += "\t";
break;
case "b":
out += "\b";
break;
case "f":
out += "\f";
break;
case '"':
case "\\":
case "/":
out += ch;
break;
case "u": {
// If \uXXXX is cut off at an incremental chunk boundary, return "as far as we got":
// emitting the incomplete hex as a literal would cause a rollback once the next
// increment completes it (breaking append-only preview); the render layer falls
// back to a new line in that case.
if (i + 5 > argsJson.length) return out;
const hex = argsJson.slice(i + 1, i + 5);
if (/^[0-9a-fA-F]{4}$/.test(hex)) {
out += String.fromCharCode(Number.parseInt(hex, 16));
i += 4;
}
break;
}
default:
out += ch;
break;
}
escaped = false;
continue;
}
if (ch === "\\") {
escaped = true;
continue;
}
if (ch === '"') return out;
out += ch;
}
return out;
}
/**
* Streaming argument preview: exec_command shows `$ <cmd>` once cmd can be read; input_command /
* input_subagent show `⌨ <name> → <id>` once the target id is available, with a non-empty
* chars / prompt appended as `<< <content>` (an empty payload just means polling, left as-is);
* run_subagent shows `run_subagent << <prompt>` once prompt can be read; other tools fall back
* to name(args-prefix).
*/
export function renderPartialToolCall(name: string, argsJson: string): string | null {
if (!argsJson) return null;
if (name === "exec_command") {
const cmd = extractPartialStringField(argsJson, "cmd");
if (cmd !== null) return `$ ${toSingleLine(cmd)}`;
return null;
}
if (name === "run_subagent") {
const prompt = extractPartialStringField(argsJson, "prompt");
if (prompt !== null) return `run_subagent << ${capPreview(toSingleLine(prompt))}`;
return null;
}
if (name === "input_command") {
const pid = extractPartialStringField(argsJson, "process_id");
if (pid === null) return null;
const chars = extractPartialStringField(argsJson, "chars");
const payload = chars ? ` << ${capPreview(visualizeControlChars(chars))}` : "";
return `⌨ input_command → ${toSingleLine(pid)}${payload}`;
}
if (name === "input_subagent") {
const sid = extractPartialStringField(argsJson, "subagent_id");
if (sid === null) return null;
const prompt = extractPartialStringField(argsJson, "prompt");
const payload = prompt ? ` << ${capPreview(toSingleLine(prompt))}` : "";
return `⌨ input_subagent → ${toSingleLine(sid)}${payload}`;
}
return `${name || "tool_call"}(${toSingleLine(argsJson)}`;
}
+162
View File
@@ -0,0 +1,162 @@
import { describe, expect, it } from "vitest";
import { Readable, Writable } from "node:stream";
import { toolCall } from "@prismshadow/penguin-core";
import type { OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
import { makeApprove, promptApproval, resolveApprovalMode } from "../src/approval.js";
import { getMessages } from "../src/i18n.js";
const t = getMessages("en");
/** An in-memory writable stream that collects everything written to output. */
function collector(): { stream: Writable; text: () => string } {
let buf = "";
const stream = new Writable({
write(chunk, _enc, cb) {
buf += chunk.toString();
cb();
},
});
return { stream, text: () => buf };
}
// mock_read_only_tool is only used for approval-mode tests, not a real tool; its permission is read-only ("r").
const readTool = (): OmniMessage<ToolCallPayload> =>
toolCall({ name: "mock_read_only_tool", arguments: '{"path":"a"}', toolCallId: "r1" });
const writeTool = (): OmniMessage<ToolCallPayload> =>
toolCall({ name: "exec_command", arguments: '{"cmd":"rm x"}', toolCallId: "w1" });
const perms: Record<string, "r" | "rw"> = {
mock_read_only_tool: "r",
exec_command: "rw",
};
const toolPermission = (name: string): "r" | "rw" | undefined => perms[name];
describe("promptApproval", () => {
it('returns "allow" when the user types "y"', async () => {
const { stream, text } = collector();
const decision = await promptApproval({
input: Readable.from(["y\n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
// Output is exactly the approval prompt itself — no input echo, no repeated tool-call rendering.
expect(text()).toBe("? Approve this tool call? [Y/n] ");
});
it('returns "allow" on empty input (Enter) — tool approval defaults to yes', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from(["\n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
});
it('returns "allow" for "yes" (case-insensitive, trimmed)', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from([" YES \n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
});
it('returns "deny" when the user types "n"', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from(["n\n"]),
output: stream,
t,
});
expect(decision).toBe("deny");
});
it('returns "allow" for unrelated input (tool approval defaults to yes)', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from(["maybe\n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
});
it('returns "deny" when the input stream ends (EOF) instead of hanging', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from([]),
output: stream,
t,
});
expect(decision).toBe("deny");
});
});
describe("resolveApprovalMode", () => {
it("maps --approve values; defaults to allow-all", () => {
expect(resolveApprovalMode("allow-all", t)).toBe("allow-all");
expect(resolveApprovalMode("read-only", t)).toBe("read-only");
expect(resolveApprovalMode("deny-all", t)).toBe("deny-all");
expect(resolveApprovalMode("always-ask", t)).toBe("always-ask");
expect(resolveApprovalMode("READ-ONLY", t)).toBe("read-only");
expect(resolveApprovalMode(undefined, t)).toBe("allow-all");
});
});
describe("makeApprove permission modes", () => {
it("allow-all → allows everything", async () => {
const approve = makeApprove({
mode: "allow-all",
toolPermission,
interactivePrompt: async () => "deny",
});
expect(await approve(readTool())).toBe("allow");
expect(await approve(writeTool())).toBe("allow");
});
it("deny-all → rejects everything", async () => {
const approve = makeApprove({
mode: "deny-all",
toolPermission,
interactivePrompt: async () => "allow",
});
expect(await approve(readTool())).toBe("deny");
expect(await approve(writeTool())).toBe("deny");
});
it("read-only → auto-allows read-only tools, prompts for the rest", async () => {
let prompted = 0;
const approve = makeApprove({
mode: "read-only",
toolPermission,
interactivePrompt: async () => {
prompted += 1;
return "deny";
},
});
// Read-only tools are auto-allowed without prompting.
expect(await approve(readTool())).toBe("allow");
expect(prompted).toBe(0);
// Read-write tools are handed off to the interactive prompt (denied here).
expect(await approve(writeTool())).toBe("deny");
expect(prompted).toBe(1);
});
it("always-ask → always delegates to the interactive prompt", async () => {
let prompted = 0;
const approve = makeApprove({
mode: "always-ask",
toolPermission,
interactivePrompt: async () => {
prompted += 1;
return "allow";
},
});
expect(await approve(readTool())).toBe("allow");
expect(await approve(writeTool())).toBe("allow");
expect(prompted).toBe(2);
});
});
+45
View File
@@ -0,0 +1,45 @@
import { describe, expect, it } from "vitest";
import { decideSigint } from "../src/commands/chat.js";
import { parseApprovalAnswer } from "../src/approval.js";
describe("decideSigint (Ctrl-C 行为状态机)", () => {
it("approving → deny(无论缓冲区是否有内容)", () => {
expect(decideSigint("approving", false)).toBe("deny");
expect(decideSigint("approving", true)).toBe("deny");
});
it("running → abort(中断当前 Task,不退出)", () => {
expect(decideSigint("running", false)).toBe("abort");
expect(decideSigint("running", true)).toBe("abort");
});
it("idle + 有输入 → clear(清空缓冲区)", () => {
expect(decideSigint("idle", true)).toBe("clear");
});
it("idle + 无输入 → confirm-exit(弹出 y/N 退出确认)", () => {
expect(decideSigint("idle", false)).toBe("confirm-exit");
});
it("confirming-exit → exit(确认中再次 Ctrl-C 直接退出)", () => {
expect(decideSigint("confirming-exit", false)).toBe("exit");
expect(decideSigint("confirming-exit", true)).toBe("exit");
});
});
describe("parseApprovalAnswer", () => {
it("y / yes(trim、不区分大小写)→ allow;n / no → deny", () => {
expect(parseApprovalAnswer("y")).toBe("allow");
expect(parseApprovalAnswer(" YES \n")).toBe("allow");
expect(parseApprovalAnswer("Y")).toBe("allow");
expect(parseApprovalAnswer("n")).toBe("deny");
expect(parseApprovalAnswer("NO")).toBe("deny");
});
it("空/无关输入用 fallback(缺省 deny;工具审批传 allow)", () => {
expect(parseApprovalAnswer("")).toBe("deny"); // default fallback
expect(parseApprovalAnswer("nope")).toBe("deny");
expect(parseApprovalAnswer("", "allow")).toBe("allow"); // tool approval defaults to allow
expect(parseApprovalAnswer("nope", "allow")).toBe("allow");
expect(parseApprovalAnswer("n", "allow")).toBe("deny"); // explicit n still denies
});
});
+68
View File
@@ -0,0 +1,68 @@
/**
* Unit tests for `config model list` rendering: provider and model_id are separate
* columns (stored fields as-is, with the default model marked `*` before the provider
* column; the request column was removed along with concatenated storage); vision falls
* back to the catalog matched by the (provider, model_id) pair; api_key is masked
* inline; fully empty columns are omitted automatically.
*/
import { describe, expect, it } from "vitest";
import type { ProjectConfig } from "@prismshadow/penguin-core";
import { formatModelRows } from "../src/commands/config.js";
describe("formatModelRows", () => {
const cfg: ProjectConfig = {
default_model: { provider: "anthropic", model_id: "claude-sonnet-4-6" },
models: [
{
provider: "anthropic",
model_id: "claude-sonnet-4-6",
context_window: 1000000,
pricing: { unit: "usd_per_mtok", cache_read: 0.3, cache_write: 3.75, output: 15 },
},
{
provider: "custom",
model_id: "my-proxy-model",
client_type: "openai",
vision: false,
api_key: "sk-test-abcd-1234",
},
],
};
it("provider 与 model_id 双列展示;预置模型 vision 经目录成对匹配,默认模型以 * 标记", () => {
const lines = formatModelRows(cfg);
expect(lines).toHaveLength(2);
expect(lines[0]).toMatch(/^\* anthropic\s+claude-sonnet-4-6\s+vision=Y/);
expect(lines[0]).toContain("price=0.3/3.75/15");
// The request column was removed; no <provider>/<id> concatenation appears anymore.
expect(lines[0]).not.toContain("request=");
expect(lines[0]).not.toContain("anthropic/claude-sonnet-4-6");
});
it("自定义模型 vision 按标注(显式 false 记 -);内联 api_key 掩码显示", () => {
const lines = formatModelRows(cfg);
expect(lines[1]).toMatch(/^ {2}custom\s+my-proxy-model\s+vision=-/);
expect(lines[1]).toContain("client_type=openai");
expect(lines[1]).toContain("api_key=****1234");
expect(lines[1]).not.toContain("sk-test-abcd-1234");
});
it("同名 model_id 双 provider 并存时各占一行,默认标记只落在成对命中的那行", () => {
const lines = formatModelRows({
default_model: { provider: "deepseek", model_id: "m1" },
models: [
{ provider: "deepseek", model_id: "m1" },
{ provider: "siliconflow", model_id: "m1" },
],
});
expect(lines[0]).toMatch(/^\* deepseek\s+m1\s+vision=Y/);
expect(lines[1]).toMatch(/^ {2}siliconflow\s+m1\s+vision=Y/);
});
it("无标注按「缺省=支持」记 Y;全空列省略", () => {
const lines = formatModelRows({
models: [{ provider: "custom", model_id: "m1" }],
});
expect(lines[0]).toBe(" custom m1 vision=Y api_key=-");
});
});
+301
View File
@@ -0,0 +1,301 @@
/**
* Integration tests for `penguin config model add|default|vision|list` (run through
* commander's parseAsync for the full command path): --model-id always takes the
* upstream id, paired with --provider to form a (provider, model_id) reference (add's
* --provider defaults to catalog-based inference, falling back to custom when
* inference fails; default / vision require --provider and raise an error when the
* reference isn't found in models — no string concatenation is ever performed); --root
* specifies the data root directory (takes priority over PENGUIN_HOME); persisted to a
* single hidden .project_config.toml (mode 0600, credentials inline, provider and
* model_id as separate columns); list displays provider and model_id as separate
* columns.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { Command } from "commander";
import { parse as parseToml } from "smol-toml";
import { DEFAULT_PROJECT_ID, projectConfigPath } from "@prismshadow/penguin-core";
import { registerConfigCommand } from "../src/commands/config.js";
import { getMessages } from "../src/i18n.js";
let tmpHome: string;
let tmpRoot: string;
let prevHome: string | undefined;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpHome = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-home-"));
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-root-"));
process.env.PENGUIN_HOME = tmpHome;
});
afterEach(async () => {
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
else process.env.PENGUIN_HOME = prevHome;
await fs.rm(tmpHome, { recursive: true, force: true });
await fs.rm(tmpRoot, { recursive: true, force: true });
});
interface TomlModelRef {
provider: string;
model_id: string;
}
/**
* Runs a `penguin config model …` command, capturing stdout / stderr and the exit code
* (without actually exiting the process; under exitOverride, commander usage errors —
* such as a missing required option — are thrown as a CommanderError, which is
* converted to a non-zero exit code).
*/
async function runModel(args: string[]): Promise<{ out: string; err: string; code: number }> {
const program = new Command();
program.exitOverride();
registerConfigCommand(program, getMessages("en"));
const out: string[] = [];
const err: string[] = [];
const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => {
out.push(String(chunk));
return true;
});
const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
err.push(String(chunk));
return true;
});
const prevExitCode = process.exitCode;
process.exitCode = undefined;
try {
await program.parseAsync(["node", "penguin", "config", "model", ...args]);
return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) };
} catch (e) {
const exitCode = (e as { exitCode?: number }).exitCode;
return { out: out.join(""), err: err.join(""), code: exitCode || 1 };
} finally {
outSpy.mockRestore();
errSpy.mockRestore();
process.exitCode = prevExitCode;
}
}
describe("penguin config model add/list(--root 与 provider / model_id 分列存储)", () => {
it("--root 优先于 PENGUIN_HOME:落盘到指定根目录的隐藏 .project_config.toml(0600)", async () => {
const add = await runModel([
"add",
"--model-id",
"my-own-model",
"--api-key",
"sk-root-secret-1",
"--root",
tmpRoot,
]);
expect(add.code).toBe(0);
// Catalog inference fails -> falls back to the custom group (provider is a separate field, never concatenated into the id).
expect(add.out).toContain("Added model (provider=custom, model_id=my-own-model).");
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
expect(path.basename(file)).toBe(".project_config.toml");
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
const parsed = parseToml(await fs.readFile(file, "utf8")) as {
models: Array<Record<string, unknown>>;
};
const entry = parsed.models.find(
(m) => m.provider === "custom" && m.model_id === "my-own-model",
);
expect(entry).toBeDefined();
expect(entry?.api_key).toBe("sk-root-secret-1");
// Concatenated storage id and request_model_id have been removed.
expect(entry?.request_model_id).toBeUndefined();
// The root directory pointed to by PENGUIN_HOME is unaffected.
await expect(fs.access(projectConfigPath(tmpHome, DEFAULT_PROJECT_ID))).rejects.toThrow();
// list also reads --root: provider and model_id as separate columns + masked api_key (the request column has been removed).
const list = await runModel(["list", "--root", tmpRoot]);
expect(list.code).toBe(0);
const line = list.out.split("\n").find((l) => l.includes("my-own-model"));
expect(line).toMatch(/custom\s+my-own-model/);
expect(line).toContain("api_key=****et-1");
expect(list.out).not.toContain("request=");
expect(list.out).not.toContain("sk-root-secret-1");
});
it("内置目录推断分组:上游 id 命中目录时条目落该 provider;--set-default 写成对引用", async () => {
const add = await runModel([
"add",
"--model-id",
"claude-sonnet-4-6",
"--set-default",
"--root",
tmpRoot,
]);
expect(add.code).toBe(0);
expect(add.out).toContain("Updated model (provider=anthropic, model_id=claude-sonnet-4-6).");
expect(add.out).toContain("Default model: (provider=anthropic, model_id=claude-sonnet-4-6)");
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as unknown as { default_model: TomlModelRef; models: Array<Record<string, unknown>> };
expect(parsed.default_model).toEqual({
provider: "anthropic",
model_id: "claude-sonnet-4-6",
});
expect(
parsed.models.find((m) => m.provider === "anthropic" && m.model_id === "claude-sonnet-4-6"),
).toBeDefined();
});
it("--provider 显式指定分组:同名上游 id 与预置条目互不冲突(各自独立条目)", async () => {
const add = await runModel([
"add",
"--model-id",
"claude-sonnet-4-6",
"--provider",
"myproxy",
"--base-url",
"https://proxy.example/v1",
"--root",
tmpRoot,
]);
expect(add.code).toBe(0);
expect(add.out).toContain("Added model (provider=myproxy, model_id=claude-sonnet-4-6).");
const list = await runModel(["list", "--root", tmpRoot]);
const line = list.out.split("\n").find((l) => l.includes("myproxy"));
expect(line).toMatch(/myproxy\s+claude-sonnet-4-6/);
expect(line).toContain("base_url=https://proxy.example/v1");
// The pre-existing anthropic entry remains (the (provider, model_id) pair naturally disambiguates).
expect(list.out.split("\n").some((l) => /anthropic\s+claude-sonnet-4-6/.test(l))).toBe(true);
});
it("client_type 缺省按分组语义(PRN-021):custom / 自建 / 网关落 openai,一方厂商不落", async () => {
// custom (catalog inference fails) and self-hosted groups (--provider not a catalog value): default to client_type=openai.
await runModel(["add", "--model-id", "my-openai-proxy", "--root", tmpRoot]);
await runModel(["add", "--model-id", "in-house-1", "--provider", "mylab", "--root", tmpRoot]);
// A non-catalog id under a first-party vendor group: client_type is not set (AgentHub auto-routes by upstream id).
await runModel([
"add",
"--model-id",
"my-fine-tune",
"--provider",
"deepseek",
"--root",
tmpRoot,
]);
// Gateway group: openai + the gateway's endpoint base URL pre-filled.
await runModel([
"add",
"--model-id",
"acme/some-model",
"--provider",
"openrouter",
"--root",
tmpRoot,
]);
// An explicit --client-type is persisted as-is, not overridden by the default rule.
await runModel([
"add",
"--model-id",
"special-1",
"--provider",
"mylab",
"--client-type",
"verbatim-type",
"--root",
tmpRoot,
]);
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as { models: Array<Record<string, unknown>> };
const by = (p: string, id: string) =>
parsed.models.find((m) => m.provider === p && m.model_id === id)!;
expect(by("custom", "my-openai-proxy").client_type).toBe("openai");
expect(by("mylab", "in-house-1").client_type).toBe("openai");
expect(by("deepseek", "my-fine-tune").client_type).toBeUndefined();
expect(by("openrouter", "acme/some-model").client_type).toBe("openai");
expect(by("openrouter", "acme/some-model").base_url).toBe("https://openrouter.ai/api/v1");
expect(by("mylab", "special-1").client_type).toBe("verbatim-type");
});
it("model default 经 --root 指定根目录设置默认模型(--model-id 上游 id + --provider 成对)", async () => {
const set = await runModel([
"default",
"--model-id",
"deepseek-v4-flash",
"--provider",
"deepseek",
"--root",
tmpRoot,
]);
expect(set.code).toBe(0);
expect(set.out).toContain(
"Default model set to (provider=deepseek, model_id=deepseek-v4-flash).",
);
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as unknown as { default_model: TomlModelRef };
expect(parsed.default_model).toEqual({
provider: "deepseek",
model_id: "deepseek-v4-flash",
});
});
});
describe("model default/vision:--provider 必填,(provider, model_id) 成对引用", () => {
it("缺 --provider:commander 用法报错,非零退出码", async () => {
const bad = await runModel(["default", "--model-id", "deepseek-v4-flash", "--root", tmpRoot]);
expect(bad.code).not.toBe(0);
expect(bad.err).toContain("--provider");
});
it("引用落空:成对引用不在 models 中,报错带成对引用与 model list 提示", async () => {
const bad = await runModel([
"default",
"--model-id",
"no-such-model",
"--provider",
"custom",
"--root",
tmpRoot,
]);
expect(bad.code).toBe(1);
expect(bad.err).toContain("(provider=custom, model_id=no-such-model)");
expect(bad.err).toContain("penguin config model list");
// The upstream id matches a pre-existing entry but --provider names the wrong group: also not found (exact pair, no fuzzy matching).
const wrongGroup = await runModel([
"vision",
"--model-id",
"claude-sonnet-4-6",
"--provider",
"openai",
"--root",
tmpRoot,
]);
expect(wrongGroup.code).toBe(1);
expect(wrongGroup.err).toContain("(provider=openai, model_id=claude-sonnet-4-6)");
expect(wrongGroup.err).toContain("penguin config model list");
});
it("model vision 成对引用命中:设置视觉模型(落盘内联表)", async () => {
const ok = await runModel([
"vision",
"--model-id",
"claude-sonnet-4-6",
"--provider",
"anthropic",
"--root",
tmpRoot,
]);
expect(ok.code).toBe(0);
expect(ok.out).toContain(
"Vision model set to (provider=anthropic, model_id=claude-sonnet-4-6).",
);
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as unknown as { vision_model: TomlModelRef };
expect(parsed.vision_model).toEqual({
provider: "anthropic",
model_id: "claude-sonnet-4-6",
});
});
});
+141
View File
@@ -0,0 +1,141 @@
/**
* Integration tests for `penguin config vault set|list|remove` (run through commander's
* parseAsync for the full command path, with PENGUIN_HOME pointed at a temp directory):
* writes to a hidden .vault.toml (mode 0600), list masks values without leaking
* plaintext, remove raises an error on a missing key, --agent-id targets a specific
* Agent, and an invalid key name / an overlong value exit with a non-zero code.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { Command } from "commander";
import { agentVaultPath, DEFAULT_PROJECT_ID } from "@prismshadow/penguin-core";
import { registerConfigCommand } from "../src/commands/config.js";
import { getMessages } from "../src/i18n.js";
let tmpRoot: string;
let prevHome: string | undefined;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-vault-"));
process.env.PENGUIN_HOME = tmpRoot;
});
afterEach(async () => {
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
else process.env.PENGUIN_HOME = prevHome;
await fs.rm(tmpRoot, { recursive: true, force: true });
});
/** Runs a `penguin config vault …` command, capturing stdout/stderr and the exit code (without actually exiting the process). */
async function runVault(args: string[]): Promise<{ out: string; err: string; code: number }> {
const program = new Command();
program.exitOverride();
registerConfigCommand(program, getMessages("en"));
const out: string[] = [];
const err: string[] = [];
const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => {
out.push(String(chunk));
return true;
});
const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
err.push(String(chunk));
return true;
});
const prevExitCode = process.exitCode;
process.exitCode = undefined;
try {
await program.parseAsync(["node", "penguin", "config", "vault", ...args]);
return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) };
} finally {
outSpy.mockRestore();
errSpy.mockRestore();
process.exitCode = prevExitCode;
}
}
describe("penguin config vault", () => {
it("set → list(掩码)→ remove 全链路;落盘为隐藏 .vault.toml 且 0600", async () => {
const set = await runVault(["set", "--key", "MY_KEY", "--value", "vault-secret-9876"]);
expect(set.code).toBe(0);
expect(set.out).toContain("Saved vault entry MY_KEY.");
const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, "default_agent");
expect(path.basename(file)).toBe(".vault.toml");
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
expect(await fs.readFile(file, "utf8")).toContain("vault-secret-9876");
const list = await runVault(["list"]);
expect(list.code).toBe(0);
expect(list.out).toContain("MY_KEY");
expect(list.out).toContain("****9876");
// Plaintext never appears in list output.
expect(list.out).not.toContain("vault-secret-9876");
const removed = await runVault(["remove", "--key", "MY_KEY"]);
expect(removed.code).toBe(0);
expect(removed.out).toContain("Removed vault entry MY_KEY.");
const empty = await runVault(["list"]);
expect(empty.out).toContain("The vault is empty.");
});
it("--agent-id 定向到目标 Agent 的 vault,不影响 default_agent", async () => {
const set = await runVault([
"set",
"--key",
"ONLY_A",
"--value",
"va-secret-value-1",
"--agent-id",
"agent-a",
]);
expect(set.code).toBe(0);
expect(
await fs.readFile(agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, "agent-a"), "utf8"),
).toContain("ONLY_A");
const defaultList = await runVault(["list"]);
expect(defaultList.out).toContain("The vault is empty.");
});
it("--root 指定数据根目录(优先于 PENGUIN_HOME),set/list 均定向到该根目录", async () => {
const otherRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-vault-root-"));
try {
const set = await runVault([
"set",
"--key",
"ROOTED_KEY",
"--value",
"root-secret-value-1",
"--root",
otherRoot,
]);
expect(set.code).toBe(0);
expect(
await fs.readFile(agentVaultPath(otherRoot, DEFAULT_PROJECT_ID, "default_agent"), "utf8"),
).toContain("ROOTED_KEY");
// The root directory pointed to by PENGUIN_HOME is unaffected.
const defaultList = await runVault(["list"]);
expect(defaultList.out).toContain("The vault is empty.");
const rootedList = await runVault(["list", "--root", otherRoot]);
expect(rootedList.out).toContain("ROOTED_KEY");
} finally {
await fs.rm(otherRoot, { recursive: true, force: true });
}
});
it("非法键名 / 超长值以非零码退出并打印原因;remove 不存在的键报错", async () => {
const badKey = await runVault(["set", "--key", "1BAD", "--value", "v"]);
expect(badKey.code).toBe(1);
expect(badKey.err).toContain("Invalid vault key");
const tooLong = await runVault(["set", "--key", "OK_BIG", "--value", "x".repeat(8193)]);
expect(tooLong.code).toBe(1);
expect(tooLong.err).toContain("too long");
const ghost = await runVault(["remove", "--key", "GHOST"]);
expect(ghost.code).toBe(1);
expect(ghost.err).toContain("Vault entry GHOST does not exist.");
});
});
+69
View File
@@ -0,0 +1,69 @@
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import { getMessages, maskApiKey, resolveLanguage } from "../src/i18n.js";
describe("resolveLanguage (env PENGUIN_LANG, default en)", () => {
let prev: string | undefined;
beforeEach(() => {
prev = process.env.PENGUIN_LANG;
});
afterEach(() => {
if (prev === undefined) delete process.env.PENGUIN_LANG;
else process.env.PENGUIN_LANG = prev;
});
it("defaults to en when unset", () => {
delete process.env.PENGUIN_LANG;
expect(resolveLanguage()).toBe("en");
});
it("matches zh exactly (case-insensitive, trimmed)", () => {
process.env.PENGUIN_LANG = "zh";
expect(resolveLanguage()).toBe("zh");
process.env.PENGUIN_LANG = " ZH ";
expect(resolveLanguage()).toBe("zh");
});
it("falls back to en for non-exact zh prefixes and anything else", () => {
process.env.PENGUIN_LANG = "zh-CN"; // no longer prefix-matched -> en
expect(resolveLanguage()).toBe("en");
process.env.PENGUIN_LANG = "fr";
expect(resolveLanguage()).toBe("en");
process.env.PENGUIN_LANG = "en";
expect(resolveLanguage()).toBe("en");
});
});
describe("getMessages", () => {
it("provides zh and en runtime + help strings", () => {
expect(getMessages("zh").modelAdded("m", "m")).toContain("已添加");
expect(getMessages("en").modelAdded("m", "m")).toContain("Added");
expect(getMessages("zh").modelUpdated("m", "m")).toContain("已更新");
expect(getMessages("en").modelUpdated("m", "m")).toContain("Updated");
// Command/option descriptions are also localized.
expect(getMessages("zh").config.addDesc).toContain("模型");
expect(getMessages("en").config.addDesc).toContain("model");
expect(getMessages("en").run.desc).toContain("Task");
// config lang copy.
expect(getMessages("zh").config.langDesc).toContain("语言");
expect(getMessages("en").config.langDesc).toContain("language");
expect(getMessages("en").langSet("zh", "/x/.zshrc")).toContain("/x/.zshrc");
expect(getMessages("zh").langInvalid("fr")).toContain("fr");
});
it("header order is agent → workspace → model", () => {
const h = getMessages("en").header("run", "ag", "/ws", "mod");
expect(h.indexOf("agent=ag")).toBeLessThan(h.indexOf("workspace=/ws"));
expect(h.indexOf("workspace=/ws")).toBeLessThan(h.indexOf("model=mod"));
});
});
describe("maskApiKey", () => {
it("masks all but the last 4 chars", () => {
expect(maskApiKey("sk-1234567890")).toBe("****7890");
});
it("fully masks short keys (≤12 chars would leak most of the secret)", () => {
expect(maskApiKey("sk-test-1234")).toBe("***");
expect(maskApiKey("short")).toBe("***");
});
it("returns - when absent", () => {
expect(maskApiKey(undefined)).toBe("-");
});
});
+110
View File
@@ -0,0 +1,110 @@
import { describe, expect, it } from "vitest";
import {
LineComposer,
PasteFilter,
endsWithContinuation,
splitTrailingPartial,
} from "../src/input.js";
/** Feeds a series of input chunks into PasteFilter, collecting the forwarded output and paste events. */
async function runFilter(chunks: string[]): Promise<{ forwarded: string; pastes: string[] }> {
const filter = new PasteFilter();
const pastes: string[] = [];
let forwarded = "";
filter.on("data", (d: Buffer) => {
forwarded += d.toString("utf8");
});
filter.on("paste", (t: string) => pastes.push(t));
for (const c of chunks) filter.write(c);
await new Promise<void>((resolve) => {
filter.end(() => resolve());
});
return { forwarded, pastes };
}
describe("splitTrailingPartial", () => {
it("holds a trailing partial-marker prefix", () => {
expect(splitTrailingPartial("abc\x1b[200", "\x1b[200~")).toEqual({
emit: "abc",
hold: "\x1b[200",
});
});
it("holds nothing when no trailing prefix", () => {
expect(splitTrailingPartial("hello", "\x1b[200~")).toEqual({
emit: "hello",
hold: "",
});
});
});
describe("PasteFilter", () => {
it("forwards normal bytes unchanged", async () => {
const { forwarded, pastes } = await runFilter(["hello\r"]);
expect(forwarded).toBe("hello\r");
expect(pastes).toEqual([]);
});
it("strips markers and emits the pasted block (incl. newlines) as one event", async () => {
const { forwarded, pastes } = await runFilter(["\x1b[200~line1\nline2\nline3\x1b[201~"]);
expect(pastes).toEqual(["line1\nline2\nline3"]);
expect(forwarded).toBe(""); // pasted content is not forwarded to readline
});
it("keeps surrounding typed bytes and paste together in order", async () => {
const { forwarded, pastes } = await runFilter(["ab\x1b[200~PASTED\x1b[201~cd\r"]);
expect(forwarded).toBe("abcd\r");
expect(pastes).toEqual(["PASTED"]);
});
it("handles a marker split across chunks", async () => {
const { forwarded, pastes } = await runFilter(["x\x1b[20", "0~mid\x1b[201", "~y\r"]);
expect(forwarded).toBe("xy\r");
expect(pastes).toEqual(["mid"]);
});
});
describe("endsWithContinuation", () => {
it("odd trailing backslashes → continuation", () => {
expect(endsWithContinuation("foo\\")).toBe(true);
expect(endsWithContinuation("foo\\\\\\")).toBe(true);
});
it("even/none → not continuation", () => {
expect(endsWithContinuation("foo")).toBe(false);
expect(endsWithContinuation("foo\\\\")).toBe(false);
});
});
describe("LineComposer", () => {
it("single line → immediate message", () => {
const c = new LineComposer();
expect(c.pushTypedLine("hello")).toEqual({ message: "hello" });
});
it("backslash continuation joins lines with \\n", () => {
const c = new LineComposer();
expect(c.pushTypedLine("a\\")).toEqual({});
expect(c.pushTypedLine("b\\")).toEqual({});
expect(c.pushTypedLine("c")).toEqual({ message: "a\nb\nc" });
});
it("paste buffers a block, Enter on empty line sends it", () => {
const c = new LineComposer();
expect(c.pushPaste("l1\nl2\n")).toEqual({ lineCount: 2, normalized: "l1\nl2" });
expect(c.hasPending()).toBe(true);
expect(c.pushTypedLine("")).toEqual({ message: "l1\nl2" });
expect(c.hasPending()).toBe(false);
});
it("paste then typed text appends the text before sending", () => {
const c = new LineComposer();
c.pushPaste("l1\nl2");
expect(c.pushTypedLine("more")).toEqual({ message: "l1\nl2\nmore" });
});
it("reset clears pending", () => {
const c = new LineComposer();
c.pushPaste("a\nb");
c.reset();
expect(c.hasPending()).toBe(false);
});
});
+90
View File
@@ -0,0 +1,90 @@
import { mkdtemp, readFile, rm } from "node:fs/promises";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { afterEach, describe, expect, it } from "vitest";
import { applyLanguageToRc, resolveShellRc, upsertBlock } from "../src/lang-config.js";
describe("resolveShellRc", () => {
it("maps zsh / bash / fish to their startup files and syntax", () => {
const zsh = resolveShellRc("/bin/zsh", "/home/u");
expect(zsh.kind).toBe("zsh");
expect(zsh.rcPath).toBe("/home/u/.zshrc");
expect(zsh.body("zh")).toBe("export PENGUIN_LANG=zh");
const bash = resolveShellRc("/usr/bin/bash", "/home/u");
expect(bash.kind).toBe("bash");
expect(bash.rcPath).toBe("/home/u/.bashrc");
const fish = resolveShellRc("/usr/local/bin/fish", "/home/u");
expect(fish.kind).toBe("fish");
expect(fish.rcPath).toBe("/home/u/.config/fish/config.fish");
expect(fish.body("en")).toBe("set -gx PENGUIN_LANG en");
});
it("falls back to ~/.profile for an unknown shell", () => {
const rc = resolveShellRc(undefined, "/home/u");
expect(rc.kind).toBe("unknown");
expect(rc.rcPath).toBe("/home/u/.profile");
});
});
describe("upsertBlock", () => {
it("appends a marked block when none exists", () => {
const out = upsertBlock("export PATH=/x\n", "export PENGUIN_LANG=zh");
expect(out).toContain("export PATH=/x");
expect(out).toContain("# >>> PenguinHarness PENGUIN_LANG >>>");
expect(out).toContain("export PENGUIN_LANG=zh");
expect(out).toContain("# <<< PenguinHarness PENGUIN_LANG <<<");
});
it("replaces the block in place and is idempotent", () => {
const first = upsertBlock("", "export PENGUIN_LANG=zh");
const second = upsertBlock(first, "export PENGUIN_LANG=en");
// Only one block remains, with its content replaced by the latest value.
expect(second.match(/PenguinHarness PENGUIN_LANG/g)?.length).toBe(2); // begin + end markers
expect(second).toContain("export PENGUIN_LANG=en");
expect(second).not.toContain("export PENGUIN_LANG=zh");
// Writing the same value again is stable (the block does not keep growing).
const third = upsertBlock(second, "export PENGUIN_LANG=en");
expect(third).toBe(second);
});
it("preserves surrounding content when replacing", () => {
const base = "line1\n" + upsertBlock("", "export PENGUIN_LANG=zh") + "line2\n";
const out = upsertBlock(base, "export PENGUIN_LANG=en");
expect(out.startsWith("line1\n")).toBe(true);
expect(out.endsWith("line2\n")).toBe(true);
expect(out).toContain("export PENGUIN_LANG=en");
});
});
describe("applyLanguageToRc", () => {
let home: string;
afterEach(async () => {
await rm(home, { recursive: true, force: true });
});
it("writes the export line to the resolved startup file", async () => {
home = await mkdtemp(join(tmpdir(), "penguin-lang-"));
const { rcPath, kind } = await applyLanguageToRc("zh", { shell: "/bin/zsh", home });
expect(kind).toBe("zsh");
expect(rcPath).toBe(join(home, ".zshrc"));
const content = await readFile(rcPath, "utf8");
expect(content).toContain("export PENGUIN_LANG=zh");
// Switching the language again updates the file in place instead of appending.
await applyLanguageToRc("en", { shell: "/bin/zsh", home });
const updated = await readFile(rcPath, "utf8");
expect(updated).toContain("export PENGUIN_LANG=en");
expect(updated).not.toContain("export PENGUIN_LANG=zh");
expect(updated.match(/# >>> PenguinHarness/g)?.length).toBe(1);
});
it("creates nested config dir for fish", async () => {
home = await mkdtemp(join(tmpdir(), "penguin-lang-"));
const { rcPath } = await applyLanguageToRc("en", { shell: "/usr/bin/fish", home });
expect(rcPath).toBe(join(home, ".config", "fish", "config.fish"));
const content = await readFile(rcPath, "utf8");
expect(content).toContain("set -gx PENGUIN_LANG en");
});
});
+716
View File
@@ -0,0 +1,716 @@
import { describe, expect, it } from "vitest";
import { Writable } from "node:stream";
import {
approvalDecision,
abortEvent,
assistantText,
compactionBegin,
compactionEnd,
requestBegin,
requestEnd,
thinkingMessage,
toolCall,
toolCallOutput,
tokenUsage,
sessionMeta,
partialText,
partialThinking,
partialToolCall,
partialToolCallOutput,
withOrigin,
} from "@prismshadow/penguin-core";
import type { MessageOrigin } from "@prismshadow/penguin-core";
import { StreamRenderer, formatAbort, humanizeTokens, renderHistory } from "../src/render.js";
import { getMessages } from "../src/i18n.js";
const t = getMessages("en");
function collector(): { stream: Writable; text: () => string } {
let buf = "";
const stream = new Writable({
write(chunk, _enc, cb) {
buf += chunk.toString();
cb();
},
});
return { stream, text: () => buf };
}
function stripAnsi(s: string): string {
// eslint-disable-next-line no-control-regex
return s.replace(/\x1b\[[0-9;]*[A-Za-z]/g, "");
}
/** Overrides a message's timestamp (the constructor defaults to the current time). */
function at<M extends { timestamp: string }>(ts: string, msg: M): M {
return { ...msg, timestamp: ts };
}
/** token_usage shorthand: request.total = req, session.total = sess (all buckets zero, sufficient for this test group). */
function usage(req: number, sess: number) {
return tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: sess },
{ cache_read: 0, cache_write: 0, output: 0, total: req },
);
}
describe("humanizeTokens", () => {
it("abbreviates with k / M and trims .0", () => {
expect(humanizeTokens(0)).toBe("0");
expect(humanizeTokens(999)).toBe("999");
expect(humanizeTokens(1000)).toBe("1k");
expect(humanizeTokens(1234)).toBe("1.2k");
expect(humanizeTokens(32000)).toBe("32k");
expect(humanizeTokens(1_500_000)).toBe("1.5M");
});
});
describe("pure formatters", () => {
it("formatAbort includes the reason", () => {
expect(stripAnsi(formatAbort({ type: "abort", reason: "ctrl-c" }, t))).toContain("ctrl-c");
});
it("renderHistory includes abort events from resumed sessions", () => {
const { stream, text } = collector();
renderHistory([assistantText("partial", "aborted"), abortEvent("aborted by user")], stream, t);
expect(stripAnsi(text())).toBe("partial [aborted]\n[abort]: aborted by user\n");
});
});
describe("StreamRenderer", () => {
it("streams partial_text deltas and does NOT re-render the complete text", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialText("start", "Hel"));
r.handle(partialText("delta", "lo "));
r.handle(partialText("delta", "world"));
r.handle(partialText("stop", "", "completed"));
r.handle(assistantText("Hello world")); // complete message: must not be re-rendered
expect(stripAnsi(text())).toBe("Hello world\n");
});
it("streams partial_thinking (dim) and skips the complete thinking", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialThinking("start", "think"));
r.handle(partialThinking("delta", "ing"));
r.handle(partialThinking("stop"));
r.handle(thinkingMessage("thinking")); // must not be re-rendered
expect(stripAnsi(text())).toBe("thinking\n");
});
it("does not render a complete tool_call without partials", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c2" }));
expect(text()).toBe("");
});
it("streams partial_tool_call with a pairing tag and skips the complete tool_call", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c4" }));
r.handle(
partialToolCall({ eventType: "delta", name: "", arguments: '{"cmd":"l', toolCallId: "c4" }),
);
r.handle(partialToolCall({ eventType: "delta", name: "", arguments: 's"}', toolCallId: "c4" }));
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c4" }));
r.handle(toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c4" }));
// The call line carries a [tool-<last-3-chars-of-id>] pairing tag matching the output line.
expect(stripAnsi(text())).toBe("[tool-c4] $ ls\n");
});
it("streams partial_tool_call_output with a tagged gutter and skips the complete tool_call_output", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCallOutput({ eventType: "start", toolCallId: "c3" }));
r.handle(partialToolCallOutput({ eventType: "delta", output: "line1\n", toolCallId: "c3" }));
r.handle(partialToolCallOutput({ eventType: "delta", output: "line2", toolCallId: "c3" }));
r.handle(partialToolCallOutput({ eventType: "stop", toolCallId: "c3" }));
r.handle(toolCallOutput({ output: "line1\nline2", toolCallId: "c3" })); // must not be re-rendered
// Each line starts with a tagged gutter (no indent) matching the call line.
expect(stripAnsi(text())).toBe("[tool-c3] >> line1\n[tool-c3] >> line2\n");
});
it("prints the retry line only when the retry request actually begins", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(requestBegin());
r.handle(requestEnd("malformed"));
expect(stripAnsi(text())).toBe(""); // the failure itself prints nothing; only the retry's start does
r.handle(requestBegin()); // retry #1 begins
expect(stripAnsi(text())).toContain("retry #1");
r.handle(requestEnd("timeout"));
r.handle(requestBegin()); // retry #2 begins
expect(stripAnsi(text())).toContain("retry #2");
// Retry #2 fails again and retries are exhausted: no next request_begin, only abort — no retry #3 appears.
r.handle(requestEnd("malformed"));
r.handle(abortEvent("malformed response failed after 2 retries"));
expect(stripAnsi(text())).not.toContain("retry #3");
// The first request of the next run is not a retry, so it prints nothing; a new failure after it counts from 1 again.
r.handle(requestBegin());
r.handle(requestEnd("timeout"));
r.handle(requestBegin());
const lines = stripAnsi(text());
expect(lines.match(/retry #1/g)).toHaveLength(2);
expect(lines).not.toContain("retry #3");
});
it("locks the screen to one streaming tool output; other messages queue until its stop", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCallOutput({ eventType: "start", toolCallId: "tA" }));
r.handle(partialToolCallOutput({ eventType: "delta", output: "a1\n", toolCallId: "tA" }));
// The screen is locked by tA: other streaming messages queue up.
r.handle(partialText("start", ""));
r.handle(partialText("delta", "hello"));
r.handle(partialToolCallOutput({ eventType: "delta", output: "a2\n", toolCallId: "tA" }));
expect(stripAnsi(text())).toBe("[tool-tA] >> a1\n[tool-tA] >> a2\n"); // hello is still queued
r.handle(partialToolCallOutput({ eventType: "stop", toolCallId: "tA" }));
r.handle(partialText("stop", "", "completed"));
expect(stripAnsi(text())).toBe("[tool-tA] >> a1\n[tool-tA] >> a2\nhello\n");
});
it("queues everything while a user prompt is active and flushes after it ends", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.beginUserPrompt();
r.handle(partialText("start", ""));
r.handle(partialText("delta", "after prompt"));
r.handle(partialText("stop", "", "completed"));
expect(text()).toBe(""); // the screen is locked while waiting for user input
r.endUserPrompt();
expect(stripAnsi(text())).toBe("after prompt\n");
});
it("does not print token_usage per turn; endTask prints [stats] line with per-task deltas", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(
sessionMeta({
session_id: "s",
provider: "custom",
model_id: "m",
model_context_window: 1,
system_prompt: "sp",
tools: [{ name: "exec_command", description: "test tool" }],
thinking_level: "medium",
agent_state: "/a",
workspace: "/w",
}),
);
// Two turns: request total 1500, 4000. Per-task token delta = 5500; session cumulative = 12000;
// context = the latest request's input+output (= total) = 4000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 200, total: 1500 },
),
);
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 12000 },
{ cache_read: 0, cache_write: 0, output: 300, total: 4000 },
),
);
expect(stripAnsi(text())).toBe(""); // no stats line is printed mid-turn
r.endTask(2345);
// Exact full-line assertion: context 4k (the latest request's total) and its delta, cumulative tokens 12k,
// per-task delta 5.5k (1500 + 4000), elapsed 2.3s (first task: session equals the delta);
// this also implies session_meta is not rendered (no /w or similar field appears in the output).
expect(stripAnsi(text())).toBe(
"[stats] context 4k (+4k) · tokens 12k (+5.5k) · 2.3s (+2.3s)\n",
);
});
it("accumulates session elapsed across tasks; context delta is vs previous task", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// task 1: context 4000, elapsed 2000ms.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 4000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 4000 },
),
);
r.endTask(2000);
// task 2: context 7000 (+3000 vs. the previous task), session elapsed cumulative 5000ms (this task +3000ms).
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 11000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 7000 },
),
);
r.endTask(3000);
const lines = stripAnsi(text()).trim().split("\n");
const last = lines[lines.length - 1]!;
// Exact full-line assertion: context 7k (delta = 7000 - 4000), cumulative session tokens 11k,
// per-task token delta 7k, total session elapsed 5s (this task +3s).
expect(last).toBe("[stats] context 7k (+3k) · tokens 11k (+7k) · 5s (+3s)");
});
it("context delta goes negative after compaction shrinks the context (no clamping)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// task 1: context 7000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 7000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 7000 },
),
);
r.endTask(1000);
// task 2: context drops to 2000 after compaction -> delta is negative (2000 - 7000 = -5k), not clamped to non-negative.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 9000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 2000 },
),
);
r.endTask(1000);
const lines = stripAnsi(text()).trim().split("\n");
expect(lines[lines.length - 1]).toBe("[stats] context 2k (-5k) · tokens 9k (+2k) · 2s (+1s)");
});
it("renders mode-specific compaction messages (summarize vs discard)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(compactionBegin({ reason: "context", mode: "summarize", context: 150, turns: 3 }));
r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "completed" }));
r.handle(compactionBegin({ reason: "manual", mode: "discard", context: 10, turns: 1 }));
r.handle(compactionEnd({ reason: "manual", mode: "discard", status: "completed" }));
r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "failed" }));
expect(stripAnsi(text())).toBe(
[
"[compaction] summarizing context (context)…",
"[compaction] done; continuing with the summarized context",
"[compaction] discarding context (manual)…",
"[compaction] done; old context discarded",
"[compaction] failed; keeping the current context",
"",
].join("\n"),
);
});
it("轮结束后的压缩:压缩完成行展示本次消耗,但不计入本轮统计增量;不更新上下文", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// Ordinary request: context 5000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 5000 },
),
);
// The compaction request's usage sits between the paired compaction events: no ordinary request_end
// follows it in this turn -> compaction after the turn has ended.
r.handle(compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }));
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 14000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 6000 },
),
);
r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "completed" }));
r.endTask(1000);
const s = stripAnsi(text());
// The compaction-done line still shows this call's usage: session cumulative 14k + this compaction's 6k.
expect(s).toContain(
"[compaction] done; continuing with the summarized context · tokens 14k (+6k)",
);
// Stats line: context stays at the ordinary-request figure of 5k; cumulative tokens 14k (includes
// compaction, following the provider), but this turn's **delta** is only the ordinary request's 5k —
// compaction after the turn ends is not attributed to this turn.
expect(s).toContain("context 5k");
expect(s).toContain("tokens 14k (+5k)");
});
it("轮途中的压缩(其后还有普通 request_end):用时含压缩跨度,Token 增量计入压缩", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// own1: ordinary request, request 5000, 00:00 -> 00:02.
r.handle(at("2026-07-05T00:00:00.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:01.000Z", usage(5000, 5000)));
r.handle(at("2026-07-05T00:00:02.000Z", requestEnd("completed")));
// Mid-turn compaction: 00:03 -> 00:13, request 6000 (the compaction's own summarization request).
r.handle(
at(
"2026-07-05T00:00:03.000Z",
compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }),
),
);
r.handle(at("2026-07-05T00:00:04.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:10.000Z", usage(6000, 14000)));
r.handle(at("2026-07-05T00:00:12.000Z", requestEnd("completed")));
r.handle(
at(
"2026-07-05T00:00:13.000Z",
compactionEnd({ reason: "context", mode: "summarize", status: "completed" }),
),
);
// The turn continues after compaction (carry-over): own2 request 2000, final request_end at 00:16 -> settles the compaction usage.
r.handle(at("2026-07-05T00:00:14.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:15.000Z", usage(2000, 16000)));
r.handle(at("2026-07-05T00:00:16.000Z", requestEnd("completed")));
r.endTask(999); // the passed-in wall clock is ignored: with a request_end present, elapsed comes from the timestamp span
const s = stripAnsi(text());
// Elapsed = first event 00:00 -> the last non-compaction request_end 00:16 = 16s (includes the 10s of
// compaction in the middle, which occupied this turn's wall clock).
// Token delta = own1 5000 + own2 2000 + compaction 6000 = 13k; context uses the ordinary-request figure after compaction, 2k.
expect(s).toContain("context 2k");
expect(s).toContain("tokens 16k (+13k)");
expect(s).toContain("16s (+16s)");
});
it("轮结束后的压缩(带 request 事件):用时止于压缩前的最后一个 request_end", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// own1: 00:00 -> 00:03.
r.handle(at("2026-07-05T00:00:00.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:01.000Z", usage(5000, 5000)));
r.handle(at("2026-07-05T00:00:03.000Z", requestEnd("completed")));
// Trailing compaction: 00:04 -> 00:24, a full 20s, with no ordinary request_end for this turn after it.
r.handle(
at(
"2026-07-05T00:00:04.000Z",
compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }),
),
);
r.handle(at("2026-07-05T00:00:05.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:20.000Z", usage(6000, 14000)));
r.handle(at("2026-07-05T00:00:23.000Z", requestEnd("completed")));
r.handle(
at(
"2026-07-05T00:00:24.000Z",
compactionEnd({ reason: "context", mode: "summarize", status: "completed" }),
),
);
r.endTask(999);
const s = stripAnsi(text());
// Elapsed = 00:00 -> the last non-compaction request_end before compaction, 00:03 = 3s (the whole 20s
// compaction span comes after it and does not count).
// Token delta is only own1's 5k; compaction's 6k is not attributed to this turn.
expect(s).toContain("tokens 14k (+5k)");
expect(s).toContain("3s (+3s)");
});
it("renders approval_decision events (approved / denied)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(approvalDecision("allow", "c1"));
r.handle(approvalDecision("deny", "c2"));
const s = stripAnsi(text());
expect(s).toContain("[approved]");
expect(s).toContain("[denied]");
});
it("keeps call → decision contiguous at prompt time and dedupes the late approval_decision event", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const tc = toolCall({ name: "exec_command", arguments: '{"cmd":"pwd"}', toolCallId: "p8" });
// Interactive approval: while locked, renders "call line -> (prompt, written directly by readline) -> result" as three contiguous lines.
r.beginUserPrompt(tc);
r.noteApprovalDecision(tc, "allow");
r.endUserPrompt();
expect(stripAnsi(text())).toBe("[tool-p8] $ pwd\n✓ [approved]\n");
// A late approval_decision event is deduped by key and not re-rendered.
r.handle(approvalDecision("allow", "p8"));
expect(stripAnsi(text())).toBe("[tool-p8] $ pwd\n✓ [approved]\n");
});
it("re-renders a half-streamed call line at approval and suppresses its late tail deltas", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const tc = toolCall({
name: "exec_command",
arguments: '{"cmd":"git status"}',
toolCallId: "h7",
});
// The call line is still mid-stream (only half its arguments rendered) when approval begins.
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "h7" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"git st',
toolCallId: "h7",
}),
);
r.beginUserPrompt(tc);
// The trailing delta / stop arrive queued while the screen is locked.
r.handle(
partialToolCall({ eventType: "delta", name: "", arguments: 'atus"}', toolCallId: "h7" }),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "h7" }));
r.noteApprovalDecision(tc, "allow");
r.endUserPrompt();
const s = stripAnsi(text());
// At approval time, the full call line is re-rendered in place from the complete message, right next to
// the result; after unlocking, the late tail is deduped and must not start a duplicate call line after
// the result line.
expect(s).toContain("[tool-h7] $ git status\n✓ [approved]\n");
expect(s.slice(s.indexOf("[approved]"))).not.toContain("[tool-h7]");
});
it("defers another call's auto-approval rendering while an interactive prompt is active", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const parent = toolCall({
name: "exec_command",
arguments: '{"cmd":"pwd"}',
toolCallId: "pa1",
});
const child = withOrigin(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "ch2" }),
"sess_kid",
);
r.beginUserPrompt(parent); // parent call's interactive prompt: locks the screen
r.noteApprovalDecision(child, "allow"); // concurrent subagent auto-approval: deferred, not inserted mid-prompt
expect(stripAnsi(text())).not.toContain("ch2");
r.noteApprovalDecision(parent, "allow"); // the prompt owner's result renders in place as usual
r.endUserPrompt();
const s = stripAnsi(text());
// Order: parent call line -> parent result -> child call line -> child result.
const iParentOk = s.indexOf("[approved]");
const iChildCall = s.indexOf("[agent-kid-tool-ch2]");
expect(s.indexOf("[tool-pa1]")).toBeGreaterThanOrEqual(0);
expect(iChildCall).toBeGreaterThan(iParentOk);
expect(s.indexOf("[approved]", iChildCall)).toBeGreaterThan(iChildCall);
});
it("endCompact settles manual /compact usage so the next task's delta excludes it", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 5000 },
),
);
r.endTask(1000);
// Manual /compact: the compaction request consumes 6000 (already shown on the compaction-done line), endCompact settles it.
r.handle(compactionBegin({ reason: "manual", mode: "summarize", context: 5000, turns: 1 }));
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 14000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 6000 },
),
);
r.handle(compactionEnd({ reason: "manual", mode: "summarize", status: "completed" }));
r.endCompact(500);
// The next task consumes only 1000: its delta must not include compaction's 6000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 15000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 1000 },
),
);
r.endTask(1000);
const lines = stripAnsi(text()).trim().split("\n");
expect(lines[lines.length - 1]).toContain("tokens 15k (+1k)");
});
it("re-renders the call line next to the decision when other output separated them (auto-approve)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// The call line is first rendered while streaming, then separated from the decision by other output.
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c5" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"ls"}',
toolCallId: "c5",
}),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c5" }));
r.handle(partialText("start", ""));
r.handle(partialText("delta", "hi"));
r.handle(partialText("stop", "", "completed"));
// Auto-approval: the call line is no longer adjacent -> it is re-rendered in place, with the result immediately following it as a pair.
r.noteApprovalDecision(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c5" }),
"allow",
);
expect(stripAnsi(text())).toBe("[tool-c5] $ ls\nhi\n[tool-c5] $ ls\n✓ [approved]\n");
});
it("does not re-render the call line when it is already adjacent to the decision", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c6" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"ls"}',
toolCallId: "c6",
}),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c6" }));
r.noteApprovalDecision(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c6" }),
"deny",
);
expect(stripAnsi(text())).toBe("[tool-c6] $ ls\n× [denied]\n");
});
});
describe("StreamRenderer — nested (origin-tagged) subagent messages", () => {
const hop: MessageOrigin = "sess_child";
it("renders nested tool calls with an agent-tool tag; skips nested text/thinking partials", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// Nested text/thinking is not rendered (the child's reply is shown via the parent tool's output gutter).
r.handle(withOrigin(partialText("delta", "child text"), hop));
r.handle(withOrigin(partialThinking("delta", "child think"), hop));
// A nested complete tool_call renders one line (so the user can see what tool the subagent is calling
// before approval); the tag is agent-<last-3-chars-of-child-session>-tool-<last-3-chars-of-id>; the
// approval line carries no tag.
r.handle(
withOrigin(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "cc1" }),
hop,
),
);
r.handle(withOrigin(approvalDecision("allow", "cc1"), hop));
expect(stripAnsi(text())).toBe("[agent-ild-tool-cc1] $ ls\n✓ [approved]\n");
});
it("renders the pending nested tool call at approval time when its stream copy has not arrived; dedupes the late copy", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const tc = withOrigin(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "cc9" }),
hop,
);
// The approval callback arrives before the forwarded message: beginUserPrompt renders the call line directly from the complete message.
r.beginUserPrompt(tc);
expect(stripAnsi(text())).toBe("[agent-ild-tool-cc9] $ ls\n");
r.endUserPrompt();
// The late forwarded copy is deduped by key and not re-rendered.
r.handle(tc);
expect(stripAnsi(text())).toBe("[agent-ild-tool-cc9] $ ls\n");
});
it("renders the pending parent tool call at approval time and suppresses its late partial stream", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.beginUserPrompt(
toolCall({ name: "exec_command", arguments: '{"cmd":"pwd"}', toolCallId: "p7" }),
);
r.endUserPrompt();
// The whole late streaming copy is deduped and skipped.
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "p7" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"pwd"}',
toolCallId: "p7",
}),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "p7" }));
expect(stripAnsi(text())).toBe("[tool-p7] $ pwd\n");
});
it("adds nested token_usage request totals to the task delta and the session total", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// One parent-session request: 1500; one child-session request: 2000 -> per-task delta 3.5k;
// session cumulative = parent 8000 + child 2000 = 10k (delta and cumulative use the same basis: parent + child).
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 200, total: 1500 },
),
);
r.handle(
withOrigin(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 2000 },
{ cache_read: 0, cache_write: 0, output: 100, total: 2000 },
),
hop,
),
);
r.endTask(1000);
const s1 = stripAnsi(text());
expect(s1).toContain("3.5k"); // the per-task delta includes child-session usage
expect(s1).toContain("10k"); // the session cumulative includes child-session usage
// The child session's cumulative persists across tasks: the next task consumes only from the parent session, cumulative = 9000 + 2000 = 11k (+1k).
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 9000 },
{ cache_read: 0, cache_write: 0, output: 100, total: 1000 },
),
);
r.endTask(1000);
const lines = stripAnsi(text()).trim().split("\n");
const last = lines[lines.length - 1]!;
expect(last).toContain("11k");
expect(last).toContain("+1k");
});
it("prints stats when a task only has nested (subagent) token usage", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(
withOrigin(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 2000 },
{ cache_read: 0, cache_write: 0, output: 100, total: 2000 },
),
hop,
),
);
r.endTask(1000);
const s = stripAnsi(text());
expect(s).toContain("[stats]");
expect(s).toContain("2k (+2k)");
});
});
describe("renderHistory (resume)", () => {
it("renders complete messages statically with interruption markers", async () => {
const { renderHistory } = await import("../src/render.js");
const { userText } = await import("@prismshadow/penguin-core");
const { stream, text } = collector();
renderHistory(
[
userText("hello"),
thinkingMessage("pondering"),
assistantText("hi there"),
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "call_653" }),
toolCallOutput({ output: "a.txt\nb.txt", toolCallId: "call_653" }),
assistantText("half answer", "aborted"),
],
stream,
);
const s = stripAnsi(text());
expect(s).toContain("> hello");
expect(s).toContain("pondering");
expect(s).toContain("hi there");
expect(s).toContain("[tool-653] $ ls");
expect(s).toContain("[tool-653] >> a.txt");
expect(s).toContain("[tool-653] >> b.txt");
// An interrupted message carries a marker (rendering includes the interrupted turn).
expect(s).toContain("half answer [aborted]");
});
it("skips events and renders nothing for empty history", async () => {
const { renderHistory } = await import("../src/render.js");
const { stream, text } = collector();
renderHistory(
[
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 1 },
{ cache_read: 0, cache_write: 0, output: 0, total: 1 },
),
],
stream,
);
expect(text()).toBe("");
});
});
+72
View File
@@ -0,0 +1,72 @@
import { describe, expect, it } from "vitest";
import { Command } from "commander";
import {
DEFAULT_HOST,
DEFAULT_PORT,
browserCommand,
browserUrl,
registerServeCommands,
resolvePort,
} from "../src/commands/serve.js";
import { getMessages } from "../src/i18n.js";
describe("resolvePort(选项 > 环境变量 > 缺省 7364)", () => {
it("都未给时用缺省 7364", () => {
expect(DEFAULT_PORT).toBe(7364);
expect(resolvePort(undefined, undefined)).toBe(7364);
expect(resolvePort(undefined, "")).toBe(7364); // an empty string counts as unset
});
it("只有环境变量时取环境变量", () => {
expect(resolvePort(undefined, "8080")).toBe(8080);
});
it("选项优先于环境变量", () => {
expect(resolvePort("9000", "8080")).toBe(9000);
});
it("非法值(非整数 / 越界)抛错", () => {
expect(() => resolvePort("abc", undefined)).toThrow(/abc/);
expect(() => resolvePort("3.14", undefined)).toThrow();
expect(() => resolvePort("-1", undefined)).toThrow();
expect(() => resolvePort("65536", undefined)).toThrow();
expect(() => resolvePort(undefined, "not-a-port")).toThrow(/not-a-port/);
});
});
describe("browserCommand(按平台选择打开命令)", () => {
const url = "http://127.0.0.1:7364/";
it("darwin → open", () => {
expect(browserCommand("darwin", url)).toEqual({ command: "open", args: [url] });
});
it("win32 → cmd /c start(空标题占位在 URL 前)", () => {
expect(browserCommand("win32", url)).toEqual({
command: "cmd",
args: ["/c", "start", "", url],
});
});
it("其他平台(linux 等)→ xdg-open", () => {
expect(browserCommand("linux", url)).toEqual({ command: "xdg-open", args: [url] });
expect(browserCommand("freebsd", url)).toEqual({ command: "xdg-open", args: [url] });
});
});
describe("browserUrl(通配监听地址转 127.0.0.1)", () => {
it("常规 host 原样拼接", () => {
expect(browserUrl(DEFAULT_HOST, 7364)).toBe("http://127.0.0.1:7364/");
expect(browserUrl("192.168.1.2", 8080)).toBe("http://192.168.1.2:8080/");
});
it("0.0.0.0 / :: 时浏览器 URL 用 127.0.0.1", () => {
expect(browserUrl("0.0.0.0", 7364)).toBe("http://127.0.0.1:7364/");
expect(browserUrl("::", 7364)).toBe("http://127.0.0.1:7364/");
});
});
describe("registerServeCommands(命令注册)", () => {
it("注册 server 与 web 两个顶层命令,web 缺省 open=true(--no-open 可关)", () => {
const program = new Command();
registerServeCommands(program, getMessages("en"));
const names = program.commands.map((c) => c.name());
expect(names).toContain("server");
expect(names).toContain("web");
const web = program.commands.find((c) => c.name() === "web")!;
expect(web.opts().open).toBe(true);
});
});
+51
View File
@@ -0,0 +1,51 @@
/**
* runTask's result reporting: when a Task ends with a main-session abort event (LLM
* failure / reconnect exhausted / user interrupt), it reports aborted=true, which
* `penguin run` maps to a non-zero exit code; a sub-session abort does not count.
*/
import { describe, expect, it } from "vitest";
import { Writable } from "node:stream";
import { abortEvent, assistantText, withOrigin } from "@prismshadow/penguin-core";
import type { OmniMessage, Session } from "@prismshadow/penguin-core";
import { StreamRenderer } from "../src/render.js";
import { runTask } from "../src/task-loop.js";
import { getMessages } from "../src/i18n.js";
const t = getMessages("en");
function fakeSession(messages: OmniMessage[]): Session {
return {
async *run() {
for (const m of messages) yield m;
},
toolPermission: () => "rw",
} as unknown as Session;
}
function silentRenderer(): StreamRenderer {
const stream = new Writable({
write(_chunk, _enc, cb) {
cb();
},
});
return new StreamRenderer(stream, t);
}
describe("runTask abort reporting", () => {
it("reports aborted=true when the task ends with a main-session abort event", async () => {
const result = await runTask(fakeSession([abortEvent("llm request error: 401")]), [], {
renderer: silentRenderer(),
t,
});
expect(result.aborted).toBe(true);
});
it("reports aborted=false on normal completion; child-session aborts do not count", async () => {
const result = await runTask(
fakeSession([withOrigin(abortEvent("child aborted"), "sess_child"), assistantText("done")]),
[],
{ renderer: silentRenderer(), t },
);
expect(result.aborted).toBe(false);
});
});
+89
View File
@@ -0,0 +1,89 @@
import { describe, expect, it } from "vitest";
import { renderPartialToolCall } from "../src/tool-render.js";
describe("renderPartialToolCall", () => {
it("renders partial exec_command args as $ <cmd-so-far>", () => {
expect(renderPartialToolCall("exec_command", '{"cmd":')).toBeNull();
expect(renderPartialToolCall("exec_command", '{"cmd":"l')).toBe("$ l");
expect(renderPartialToolCall("exec_command", '{"cmd":"ls"}')).toBe("$ ls");
expect(renderPartialToolCall("exec_command", '{"cmd":"echo \\"hi\\"')).toBe('$ echo "hi"');
});
it("renders run_subagent as run_subagent << <prompt>, folded to one line", () => {
expect(renderPartialToolCall("run_subagent", '{"prompt":')).toBeNull();
expect(renderPartialToolCall("run_subagent", '{"prompt":"analy')).toBe("run_subagent << analy");
expect(renderPartialToolCall("run_subagent", '{"prompt":"line1\\nline2"}')).toBe(
"run_subagent << line1 line2",
);
});
it("renders input_command polls (empty chars) without a payload", () => {
expect(renderPartialToolCall("input_command", '{"process_id":')).toBeNull();
expect(renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d"}')).toBe(
"⌨ input_command → proc-1a2b3c4d",
);
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":""}'),
).toBe("⌨ input_command → proc-1a2b3c4d");
});
it("renders non-empty input_command chars with visible control characters", () => {
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"y\\n"}'),
).toBe("⌨ input_command → proc-1a2b3c4d << y\\n");
// U+0003 (Ctrl-C) is rendered in caret notation.
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"\\u0003"}'),
).toBe("⌨ input_command → proc-1a2b3c4d << ^C");
// Disambiguates literal backslash escapes: chars "a", "\", "n" render as a\\n, distinct from a real newline \n.
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"a\\\\n"}'),
).toBe("⌨ input_command → proc-1a2b3c4d << a\\\\n");
});
it("keeps input_command previews append-only across \\uXXXX delta boundaries", () => {
const stages = [
'{"process_id":"proc-1a2b3c4d","chars":"y',
'{"process_id":"proc-1a2b3c4d","chars":"y\\u0',
'{"process_id":"proc-1a2b3c4d","chars":"y\\u0003',
];
const previews = stages.map((s) => renderPartialToolCall("input_command", s)!);
expect(previews[0]).toBe("⌨ input_command → proc-1a2b3c4d << y");
// An incomplete \u escape is treated as "stop here" rather than emitting the raw hex as literal text.
expect(previews[1]).toBe("⌨ input_command → proc-1a2b3c4d << y");
expect(previews[2]).toBe("⌨ input_command → proc-1a2b3c4d << y^C");
for (let i = 1; i < previews.length; i++) {
expect(previews[i]!.startsWith(previews[i - 1]!)).toBe(true);
}
});
it("renders input_subagent polls without a payload and follow-up prompts with one", () => {
expect(
renderPartialToolCall("input_subagent", '{"subagent_id":"subagent-9f8e7d6c","prompt":""}'),
).toBe("⌨ input_subagent → subagent-9f8e7d6c");
expect(
renderPartialToolCall(
"input_subagent",
'{"subagent_id":"subagent-9f8e7d6c","prompt":"continue with the tests"}',
),
).toBe("⌨ input_subagent → subagent-9f8e7d6c << continue with the tests");
});
it("truncates long payload previews and stops growing afterwards", () => {
const long = "x".repeat(130);
const capped = renderPartialToolCall(
"input_subagent",
`{"subagent_id":"subagent-9f8e7d6c","prompt":"${long}"}`,
);
expect(capped).toBe(`⌨ input_subagent → subagent-9f8e7d6c << ${"x".repeat(120)}…`);
const longer = renderPartialToolCall(
"input_subagent",
`{"subagent_id":"subagent-9f8e7d6c","prompt":"${long}yyy"}`,
);
expect(longer).toBe(capped);
});
it("falls back to name(args-prefix) for unknown tools", () => {
expect(renderPartialToolCall("search", '{"q":"hi')).toBe('search({"q":"hi');
});
});
+7
View File
@@ -0,0 +1,7 @@
{
"extends": "../../tsconfig.base.json",
"compilerOptions": {
"rootDir": "."
},
"include": ["src", "test"]
}
+22
View File
@@ -0,0 +1,22 @@
import { defineConfig } from "tsup";
export default defineConfig({
entry: ["src/index.ts"],
format: ["esm"],
target: "node20",
platform: "node",
clean: true,
sourcemap: true,
// Bundle the workspace source @prismshadow/penguin-core, but keep third-party deps (incl.
// CJS yaml / smol-toml / agenthub) external and resolved from node_modules at runtime —
// avoids bundling CJS deps into ESM and triggering a "Dynamic require" error.
// @prismshadow/penguin-skills must stay external: it reads files under its own skills/ dir
// at runtime (files are the source of truth); bundling would break paths relative to the
// package root. cli already declares it as a direct dependency.
// @prismshadow/penguin-server stays external: the penguin server / web commands import it
// dynamically at runtime.
// tsup treats this package's package.json dependencies as external by default, so these
// deps are already declared there.
noExternal: ["@prismshadow/penguin-core"],
banner: { js: "#!/usr/bin/env node" },
});