Initialize repository with harness code and assets

Initial import of all source code, config, and README assets: the
packages workspace (cli, core, server, web, docs, landing, skills),
build scripts, tooling config, and CI workflows.

Includes the data-layout revision made on this branch: the local data
root defaults to ~/.penguin/data (PENGUIN_HOME still overrides; the
installer keeps its binaries in ~/.penguin), and every Agent lives
under <project>/agents/<agent>/ — path helpers, the three
agent-enumeration scans, the system prompt, built-in Skills, tests
and docs all follow the new layout.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_018ihk8iQuo3kv2aPjAYEPuR
This commit is contained in:
Yaowei Zheng
2026-07-19 14:06:53 +08:00
committed by GitHub
parent 056bed7aeb
commit 45bfae6e94
543 changed files with 92949 additions and 0 deletions
+9
View File
@@ -0,0 +1,9 @@
# Copy to .env (ignored by .gitignore) and fill in real values for local runs and e2e.
# e2e picks a Provider by available key, in order: Claude (ANTHROPIC_API_KEY) -> DeepSeek (DEEPSEEK_API_KEY).
ANTHROPIC_API_KEY=
# Optional: custom Claude gateway URL (defaults to AgentHub if unset).
# ANTHROPIC_BASE_URL=
# DeepSeek (CI's e2e uses this key, model deepseek-v4-flash).
DEEPSEEK_API_KEY=
# Optional: custom DeepSeek gateway URL (defaults to the official https://api.deepseek.com).
# DEEPSEEK_BASE_URL=
+57
View File
@@ -0,0 +1,57 @@
# CI: build -> style (Prettier) -> typecheck (tsc) -> unit tests (vitest) -> live e2e (DeepSeek).
# Build first: core's exports point at dist/, and cli's type resolution and runtime imports both need core's build output.
# e2e needs the repo secret DEEPSEEK_API_KEY; when absent (e.g. forks) that step self-skips and the other checks run as usual.
name: CI
# Limit triggers to avoid duplicate runs: push runs only on main/dev; PRs always run once (no target-branch filter --
# this repo's PRs often target integration branches rather than main/dev, and a target filter would leave them with no CI).
on:
push:
branches: [main, dev]
pull_request:
workflow_dispatch:
concurrency:
group: ci-${{ github.workflow }}-${{ github.event_name }}-${{ github.ref }}
cancel-in-progress: true
jobs:
ci:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
# pnpm version comes from package.json's packageManager field.
- uses: pnpm/action-setup@v4
- uses: actions/setup-node@v5
with:
node-version: 24
cache: pnpm
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Build (tsup)
run: pnpm build
- name: Code style (Prettier)
run: pnpm format:check
- name: Typecheck (tsc)
run: pnpm typecheck
- name: Unit tests (vitest)
run: pnpm test
# The secret is exposed only in this step (step-level env); earlier steps and third-party actions can't see it.
# The secrets context can't be used in if expressions, so skip inside the shell when it's absent (e.g. forks).
- name: E2E (live LLM via DeepSeek)
env:
DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }}
run: |
if [ -z "$DEEPSEEK_API_KEY" ]; then
echo "DEEPSEEK_API_KEY not available; skipping e2e."
exit 0
fi
pnpm test:e2e
+75
View File
@@ -0,0 +1,75 @@
# Public site: build the landing page (site root) and the docs site (/docs/) with Vite
# and deploy them as ONE artifact to GitHub Pages (scripts/build-site.mjs assembles the
# tree — landing dist with the docs dist copied under docs/).
# - Pull requests touching either package only BUILD (validation) — no Pages
# configuration and no deploy, so PRs never fail on Pages availability.
# - Pushes to main (and manual dispatch) build + deploy. BASE_PATH is derived from the
# repository name so project-pages URLs (https://<owner>.github.io/<repo>/) resolve;
# repos served from a custom domain / user pages can set it to "/" instead.
# First-time setup: repository Settings -> Pages -> Source = "GitHub Actions".
name: Deploy Site
on:
push:
branches: [main]
paths:
- "packages/landing/**"
- "packages/docs/**"
- "scripts/build-site.mjs"
- ".github/workflows/pages.yml"
pull_request:
paths:
- "packages/landing/**"
- "packages/docs/**"
- "scripts/build-site.mjs"
- ".github/workflows/pages.yml"
workflow_dispatch:
permissions:
contents: read
pages: write
id-token: write
# One deploy at a time; let an in-flight production deploy finish rather than cancelling it.
concurrency:
group: pages-${{ github.event_name }}-${{ github.ref }}
cancel-in-progress: false
jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
# pnpm version comes from package.json's packageManager field.
- uses: pnpm/action-setup@v4
- uses: actions/setup-node@v5
with:
node-version: 24
cache: pnpm
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Build site (landing + docs, vite)
env:
BASE_PATH: /${{ github.event.repository.name }}/
run: pnpm build:site
# Only the deploying runs need the Pages artifact (PRs stop at the build check).
- if: github.event_name != 'pull_request'
uses: actions/upload-pages-artifact@v3
with:
path: packages/landing/dist
deploy:
if: github.event_name != 'pull_request'
needs: build
runs-on: ubuntu-latest
environment:
name: github-pages
url: ${{ steps.deployment.outputs.page_url }}
steps:
- id: deployment
uses: actions/deploy-pages@v4
+236
View File
@@ -0,0 +1,236 @@
# Release: tag v* -> build the one-line install artifacts and publish a GitHub Release.
# Two parallel jobs:
# - release: build the monorepo -> pnpm deploy a production CLI dir -> assemble penguin/ (bin + lib + web)
# -> four platform packages each bundling the official Node runtime + a universal package -> SHA256 files -> upload to the Release.
# Artifacts: penguin-{linux,darwin}-{x64,arm64}.tar.gz, penguin-universal.tar.gz,
# their .sha256 files, SHA256SUMS, and install.sh; one version per tag, multiple versions coexist.
# - publish-npm: publish the whole chain (@prismshadow/penguin-skills -> @prismshadow/penguin-core
# -> @prismshadow/penguin-server -> @prismshadow/penguin-cli) to npm at the tag version.
# skills/core serve the penguin-sdk Skill's `npm install`; server ships the built web assets inside
# the package (web-dist/, its default web dir falls back to it), so `npm install -g
# @prismshadow/penguin-cli` alone yields a working `penguin` incl. the Web UI (needs Node >= 24).
# The publish flow mirrors AgentHub's publish.yml: OIDC trusted publishing (environment: npm +
# id-token: write, no token). A Trusted Publisher can only be configured in the settings page of a
# package that ALREADY EXISTS on the registry, so a brand-new package cannot be first-published by
# this workflow. Release checklist for a new package: (1) a maintainer bootstrap-publishes it once
# manually with a one-off granular token (revoke it afterwards), running the same prepare steps as
# this job (stamp versions, build, copy LICENSE + web-dist) and publishing with `pnpm publish
# --access public --no-git-checks` -- NEVER `npm publish`, which keeps workspace:* deps unrewritten
# and yields a package that fails to install (Unsupported URL Type "workspace:");
# (2) configure this repo + workflow as its Trusted Publisher on npmjs; (3) subsequent tags publish
# via OIDC. The publish step is idempotent (versions already on the registry are skipped), so a tag
# that failed mid-chain can be re-run as-is after fixing the config.
# npm publishing lives here rather than a separate `on: release: published` workflow: the Release is created
# by this workflow's GITHUB_TOKEN, and GitHub won't trigger other workflows' release events from that.
name: Release
on:
push:
tags: ["v*"]
workflow_dispatch:
inputs:
tag:
description: "Release tag (e.g. v0.1.0)"
required: true
env:
# Bundled Node runtime version (official nodejs.org dist, aligned with engines >=24).
NODE_RUNTIME_VERSION: v24.18.0
jobs:
release:
runs-on: ubuntu-latest
permissions:
contents: write
steps:
# On manual dispatch, check out the tag itself (not the selected branch HEAD): when re-uploading an existing
# tag's artifacts this keeps them in sync with the tag's source. On tag-push, leaving ref empty is the default.
- uses: actions/checkout@v5
with:
ref: ${{ github.event_name == 'workflow_dispatch' && inputs.tag || '' }}
# pnpm version comes from package.json's packageManager field.
- uses: pnpm/action-setup@v4
- uses: actions/setup-node@v5
with:
node-version: 24
cache: pnpm
- name: Install dependencies
run: pnpm install --frozen-lockfile
# Inject the release tag into core's VERSION constant (the source for CLI --version and the install-complete
# message); otherwise artifacts always carry the in-repo dev version and multiple installs can't be told apart.
- name: Stamp release version
run: |
TAG="${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}"
V="${TAG#v}"
grep -q 'export const VERSION = "' packages/core/src/index.ts
sed -i "s/export const VERSION = \"[^\"]*\"/export const VERSION = \"$V\"/" packages/core/src/index.ts
- name: Build (tsup + vite)
run: pnpm build
# lib/: the CLI and its production deps (including workspace core/server/skills, all build outputs);
# pnpm 10's deploy needs --legacy (this repo doesn't enable inject-workspace-packages).
# bin/penguin launcher: resolve its own real path (following symlinks) -> default PENGUIN_WEB_DIST to
# the sibling web/ -> use the bundled runtime (node/bin/node) if present, else fall back to system node.
- name: Assemble penguin/ (lib + web + bin)
run: |
pnpm --filter @prismshadow/penguin-cli --prod deploy --legacy "$PWD/out/penguin/lib"
cp -r packages/web/dist out/penguin/web
mkdir -p out/penguin/bin
cat > out/penguin/bin/penguin <<'EOF'
#!/bin/sh
SELF="$0"
while [ -h "$SELF" ]; do
DIR="$(cd "$(dirname "$SELF")" && pwd)"
SELF="$(readlink "$SELF")"
case "$SELF" in /*) ;; *) SELF="$DIR/$SELF" ;; esac
done
DIR="$(cd "$(dirname "$SELF")/.." && pwd)"
export PENGUIN_WEB_DIST="${PENGUIN_WEB_DIST:-$DIR/web}"
if [ -x "$DIR/node/bin/node" ]; then
exec "$DIR/node/bin/node" "$DIR/lib/dist/index.js" "$@"
fi
exec node "$DIR/lib/dist/index.js" "$@"
EOF
chmod +x out/penguin/bin/penguin
# Platform packages: linux uses .tar.xz, darwin uses .tar.gz (nodejs.org naming);
# node/ is only lightly trimmed (drop share/doc and share/man, keep the rest).
- name: Package platform + universal tarballs
run: |
mkdir -p dist-artifacts
for target in linux-x64 linux-arm64 darwin-x64 darwin-arm64; do
os="${target%%-*}"
arch="${target#*-}"
name="node-$NODE_RUNTIME_VERSION-$os-$arch"
if [ "$os" = "linux" ]; then ext="tar.xz"; else ext="tar.gz"; fi
curl -fsSL "https://nodejs.org/dist/$NODE_RUNTIME_VERSION/$name.$ext" -o "/tmp/$name.$ext"
rm -rf /tmp/node-runtime out/penguin/node
mkdir -p /tmp/node-runtime
if [ "$ext" = "tar.xz" ]; then
tar -xJf "/tmp/$name.$ext" -C /tmp/node-runtime
else
tar -xzf "/tmp/$name.$ext" -C /tmp/node-runtime
fi
mv "/tmp/node-runtime/$name" out/penguin/node
rm -rf out/penguin/node/share/doc out/penguin/node/share/man
tar -czf "dist-artifacts/penguin-$os-$arch.tar.gz" -C out penguin
done
# Universal package: no bundled runtime, requires system Node >= 24.
rm -rf out/penguin/node
tar -czf dist-artifacts/penguin-universal.tar.gz -C out penguin
# SHA256SUMS summary + a same-named .sha256 per artifact (install.sh verifies against the latter).
- name: Generate SHA256 checksums
run: |
cd dist-artifacts
sha256sum *.tar.gz > SHA256SUMS
for f in *.tar.gz; do
sha256sum "$f" > "$f.sha256"
done
# On tag-push use the ref name; on manual dispatch use the input tag.
- name: Publish GitHub Release
uses: softprops/action-gh-release@v2
with:
tag_name: ${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}
files: |
dist-artifacts/*.tar.gz
dist-artifacts/*.sha256
dist-artifacts/SHA256SUMS
install.sh
publish-npm:
name: Publish npm packages
runs-on: ubuntu-latest
# OIDC trusted publishing (same as AgentHub's publish.yml): no token; npmjs establishes trust via
# environment `npm` + this workflow. A publish failure doesn't affect the release job (independent, parallel).
permissions:
id-token: write
contents: read
environment:
name: npm
url: https://www.npmjs.com/package/@prismshadow/penguin-cli
steps:
- uses: actions/checkout@v5
with:
ref: ${{ github.event_name == 'workflow_dispatch' && inputs.tag || '' }}
- uses: pnpm/action-setup@v4
- uses: actions/setup-node@v5
with:
node-version: 24
cache: pnpm
registry-url: "https://registry.npmjs.org"
# npm Trusted Publishing (OIDC) requires npm CLI >= 11.5.1: Node 24 ships npm 11.x, which satisfies it;
# this just asserts the version to guard against regressions -- pnpm publish's registry auth ultimately
# delegates to system npm, ordinary CI doesn't exercise OIDC (so a green run won't catch it), and too old
# a version only surfaces as an auth failure when actually publishing a tag.
- name: Assert npm supports trusted publishing (>= 11.5.1)
run: |
V="$(npm --version)"
echo "npm $V"
node -e 'const [M, m, p] = process.argv[1].split(".").map(Number); if (M < 11 || (M === 11 && (m < 5 || (m === 5 && p < 1)))) { console.error("npm " + process.argv[1] + " < 11.5.1"); process.exit(1); }' "$V"
- name: Install dependencies
run: pnpm install --frozen-lockfile
# Version always comes from the tag: package version and core's VERSION constant are injected together (the
# repo keeps the dev version). All published packages must bump in lockstep -- pnpm publish rewrites every
# workspace:* dep to the dependency's current version, so the versions must match.
- name: Stamp release version
run: |
TAG="${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}"
V="${TAG#v}"
grep -q 'export const VERSION = "' packages/core/src/index.ts
sed -i "s/export const VERSION = \"[^\"]*\"/export const VERSION = \"$V\"/" packages/core/src/index.ts
(cd packages/skills && npm version --no-git-tag-version "$V")
(cd packages/core && npm version --no-git-tag-version "$V")
(cd packages/server && npm version --no-git-tag-version "$V")
(cd packages/cli && npm version --no-git-tag-version "$V")
# Dependency order: cli bundles core's dist (tsup noExternal), server/web need core's types;
# web is built here only to be copied into the server package below.
- name: Build and test
run: |
pnpm --filter @prismshadow/penguin-skills build
pnpm --filter @prismshadow/penguin-core build
pnpm --filter @prismshadow/penguin-server build
pnpm --filter @prismshadow/penguin-web build
pnpm --filter @prismshadow/penguin-cli build
pnpm --filter @prismshadow/penguin-skills test
pnpm --filter @prismshadow/penguin-core test
pnpm --filter @prismshadow/penguin-server test
pnpm --filter @prismshadow/penguin-cli test
# Copy LICENSE into the package dirs (the files allowlist includes it, so artifacts ship the license)
# and the built web assets into the server package (web-dist/, in its files allowlist: an npm install
# serves the Web UI from there without PENGUIN_WEB_DIST).
# Idempotent: a version already on the registry is skipped. The check tests `npm view`'s output
# rather than its exit code -- for a missing version of an existing package, older npm exits 0 and
# newer npm exits 1, but the output is non-empty only when the version exists. Re-running the tag
# after a mid-chain failure (e.g. Trusted Publisher not configured yet) picks up where it left off.
- name: Publish to npm
run: |
TAG="${{ github.event_name == 'workflow_dispatch' && inputs.tag || github.ref_name }}"
V="${TAG#v}"
cp LICENSE packages/skills/LICENSE
cp LICENSE packages/core/LICENSE
cp LICENSE packages/server/LICENSE
cp LICENSE packages/cli/LICENSE
rm -rf packages/server/web-dist
cp -r packages/web/dist packages/server/web-dist
for pkg in skills core server cli; do
name="@prismshadow/penguin-$pkg"
if [ -n "$(npm view "$name@$V" version 2>/dev/null || true)" ]; then
echo "$name@$V already on the registry, skipping."
continue
fi
pnpm --filter "$name" publish --access public --no-git-checks
done
+31
View File
@@ -0,0 +1,31 @@
.DS_Store
# dependencies
node_modules/
# build output
dist/
*.tsbuildinfo
# web assets copied into the server package at npm publish time
packages/server/web-dist/
# secrets / local config
.env
.env.*
!.env.example
# agenthub trace cache
cache/
# legacy .penguin workspace symlink (no longer created; ignored for old working copies)
.penguin
# logs
*.log
# claude code session state (worktrees, skills)
.claude/
# playwright e2e artifacts
packages/web/test-results/
packages/web/playwright-report/
+12
View File
@@ -0,0 +1,12 @@
# build output and dependencies
node_modules/
dist/
pnpm-lock.yaml
# docs-only area (mostly Chinese; not machine-formatted)
specs/
*.md
# Playwright e2e artifacts
test-results/
playwright-report/
+3
View File
@@ -0,0 +1,3 @@
{
"printWidth": 100
}
+130
View File
@@ -0,0 +1,130 @@
<p align="center">
<img src="packages/landing/public/penguin-logo.svg" alt="PenguinHarness logo" width="88" />
</p>
<h1 align="center">PenguinHarness</h1>
<p align="center"><b>Efficient Self-Improving Harness for Everyone</b></p>
<p align="center">
Open-source, local-first infrastructure that builds AI agents for you —
from automatic agent construction to recursive self-improvement.
</p>
<p align="center">
<a href="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/ci.yml"><img src="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
<a href="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/pages.yml"><img src="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/pages.yml/badge.svg" alt="Deploy Site" /></a>
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache--2.0-blue" alt="License: Apache-2.0" /></a>
<img src="https://img.shields.io/badge/node-%E2%89%A5%2024-brightgreen" alt="Node >= 24" />
</p>
<p align="center">
English | <a href="README.zh.md">简体中文</a> ·
<a href="https://prism-shadow.github.io/penguin-harness/">Website</a> ·
<a href="https://prism-shadow.github.io/penguin-harness/docs/">Docs</a> ·
<a href="https://prism-shadow.github.io/penguin-harness/blog">Blog</a>
</p>
<p align="center">
<picture>
<source media="(prefers-color-scheme: dark)" srcset="packages/landing/src/assets/shots/chat-en-dark.webp" />
<img src="packages/landing/src/assets/shots/chat-en-light.webp" alt="PenguinHarness Web App — multi-session chat with live streaming tool calls" width="920" />
</picture>
</p>
---
## Why PenguinHarness
- **Simplest Is the Best** — a deliberately minimal toolset over clean low-level interfaces: fewer tool calls, fewer tokens, complex tasks done efficiently.
- **Harness for Building Agents** — with the PenguinHarness SDK, an Agent builds complete Agent applications for you, autonomously, from scratch.
- **Harness for Recursive Self-Improvement** — with PenguinHarness Skills, an Agent evaluates and optimizes itself: benchmark, find the lost points, ship version N+1, snapshot before every round.
- **Local-first and lightweight** — 100% open source, runs on a single CPU, your data never leaves the machine. 1000+ online and local models reachable through one gateway.
- **Everything observable** — every request, tool call and approval decision lands in an append-only Trace; any Session can be resumed from it.
## Quickstart
Install with one command (Linux / macOS, x64 / arm64, bundled Node runtime):
```bash
curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh
```
Or via npm (requires Node >= 24; the command it installs is `penguin`):
```bash
npm install -g @prismshadow/penguin-cli
```
Then launch the Web App — or stay in the terminal:
```bash
penguin web # start the service and open http://127.0.0.1:7364 (first login: admin / admin123)
penguin server # same service, headless
# configure a model once (or use the in-app Models page)
penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default
penguin run -m "Create hello.txt containing Hello, Penguin" # one-shot task
penguin chat # interactive REPL (/compact, /exit, Ctrl-C to interrupt)
```
Using the SDK directly:
```ts
import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core";
const agent = await createAgent({ agentId: "default_agent" });
const session = await agent.createSession({ workspaceDir: process.cwd() });
for await (const output of session.run([userText("Create hello.txt containing hi")], {
approve: async () => "allow", // per-tool-call approval
})) {
if (isCompleteModelMessage(output) && output.payload.type === "text") {
console.log(output.payload.text);
}
}
```
## What's inside
A pnpm monorepo (TypeScript, Node >= 24). One install ships four layers that share a single data directory (`~/.penguin/data`) and a single message protocol (OmniMessage):
| Package | Name | Role |
| --- | --- | --- |
| [`packages/core`](packages/core) | `@prismshadow/penguin-core` | SDK & engine: ReAct loop, OmniMessage protocol, LLM/Environment interface contracts, Agent State, Trace |
| [`packages/cli`](packages/cli) | `@prismshadow/penguin-cli` | The `penguin` command: REPL, one-shot runs, model & vault config, service launcher |
| [`packages/server`](packages/server) | `@prismshadow/penguin-server` | Web backend: HTTP API + SSE streaming, multi-user auth, Project authorization, usage stats |
| [`packages/web`](packages/web) | `@prismshadow/penguin-web` | Web App: multi-session chat, Agent/skill/model management, Trace observability, evaluation center |
| [`packages/skills`](packages/skills) | `@prismshadow/penguin-skills` | Built-in skill library (agent creation, benchmarking, evaluation, optimization, …) |
| [`packages/landing`](packages/landing) | — | Product landing page (this repo's website) |
| [`packages/docs`](packages/docs) | — | Documentation site (bilingual, deployed under `/docs/`) |
Responsibilities split by source of truth: the **SDK** owns protocol and execution (message parsing, the agent loop, tools), the **Server** owns the multi-user runtime (auth, SSE streaming, scheduled tasks), and the **file layer** under `~/.penguin/data` owns everything editable and recorded (prompts, Skills, secrets, Traces). The full design-by-design map is in [Architecture → Division of responsibilities](https://prism-shadow.github.io/penguin-harness/docs/architecture).
## Documentation
The docs site covers both usage and design: [Introduction](https://prism-shadow.github.io/penguin-harness/docs/) · [Quickstart](https://prism-shadow.github.io/penguin-harness/docs/quickstart) · [Architecture](https://prism-shadow.github.io/penguin-harness/docs/architecture) · [The OmniMessage Protocol](https://prism-shadow.github.io/penguin-harness/docs/omni-message) · [Core Interfaces](https://prism-shadow.github.io/penguin-harness/docs/interfaces) · [The Agent Loop](https://prism-shadow.github.io/penguin-harness/docs/agent-loop) · [CLI Reference](https://prism-shadow.github.io/penguin-harness/docs/cli) · [Server API](https://prism-shadow.github.io/penguin-harness/docs/server-api) · [Configuration](https://prism-shadow.github.io/penguin-harness/docs/configuration)
Every doc page has a "Copy Markdown" button, so you can paste it straight into a model context.
## Development
```bash
pnpm install
pnpm build # build first: core's exports point at dist/
pnpm typecheck
pnpm test
pnpm dev:server # backend at 127.0.0.1:7364
pnpm dev:web # web app (Vite) at 127.0.0.1:7365, /api proxied
pnpm dev:docs # docs site (Vite) at 127.0.0.1:7367
BASE_PATH=/ pnpm build:site # assemble landing + docs exactly like the Pages deploy
```
Copy `.env.example` to `.env` for model credentials in development. E2E tests run against a live model (`pnpm test:e2e`, needs `DEEPSEEK_API_KEY`).
## License
[Apache-2.0](LICENSE) © 2026 Prism Shadow
+129
View File
@@ -0,0 +1,129 @@
<p align="center">
<img src="packages/landing/public/penguin-logo.svg" alt="PenguinHarness logo" width="88" />
</p>
<h1 align="center">PenguinHarness</h1>
<p align="center"><b>Efficient Self-Improving Harness for Everyone</b></p>
<p align="center">
开源、本地优先的 AI Agent 基础设施——从自动构建 Agent 到递归自我进化。
</p>
<p align="center">
<a href="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/ci.yml"><img src="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
<a href="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/pages.yml"><img src="https://github.com/Prism-Shadow/penguin-harness/actions/workflows/pages.yml/badge.svg" alt="Deploy Site" /></a>
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache--2.0-blue" alt="License: Apache-2.0" /></a>
<img src="https://img.shields.io/badge/node-%E2%89%A5%2024-brightgreen" alt="Node >= 24" />
</p>
<p align="center">
<a href="README.md">English</a> | 简体中文 ·
<a href="https://prism-shadow.github.io/penguin-harness/">官网</a> ·
<a href="https://prism-shadow.github.io/penguin-harness/docs/">文档</a> ·
<a href="https://prism-shadow.github.io/penguin-harness/blog">博客</a>
</p>
<p align="center">
<picture>
<source media="(prefers-color-scheme: dark)" srcset="packages/landing/src/assets/shots/chat-zh-dark.webp" />
<img src="packages/landing/src/assets/shots/chat-zh-light.webp" alt="PenguinHarness Web App——多 Session 对话与实时流式工具调用" width="920" />
</picture>
</p>
---
## 为什么选择 PenguinHarness
- **Simplest Is the Best**——在干净的底层接口之上刻意保持极简的工具集:更少的工具调用、更少的 Token,高效完成复杂任务。
- **Harness for Building Agents**——基于 PenguinHarness SDK,由一个 Agent 从零开始为你自主构建完整的 Agent 应用。
- **Harness for Recursive Self-Improvement**——基于 PenguinHarness Skills,Agent 评估并优化自己:跑 Benchmark、找失分点、产出 N+1 版本,每轮之前先做快照。
- **本地优先且轻量**——100% 开源,一颗 CPU 即可运行,数据不出机器;经统一网关可接入 1000+ 在线与本地模型。
- **全量可观测**——每次请求、工具调用与审批决策都以追加方式写入 Trace,任何 Session 均可从 Trace 恢复。
## 快速开始
一行命令安装(Linux / macOS,x64 / arm64,内嵌 Node 运行时,解压即用):
```bash
curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh
```
或经 npm 安装(需系统 Node >= 24,安装后的命令为 `penguin`):
```bash
npm install -g @prismshadow/penguin-cli
```
然后启动 Web App,或直接留在终端:
```bash
penguin web # 启动服务并打开 http://127.0.0.1:7364(初始账号 admin / admin123)
penguin server # 同一服务,无头运行
# 先配置一次模型(也可在 Web 的模型页完成)
penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default
penguin run -m "创建 hello.txt,内容为 Hello, Penguin" # 单次任务
penguin chat # 交互式 REPL(/compact、/exit,Ctrl-C 中断)
```
直接使用 SDK:
```ts
import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core";
const agent = await createAgent({ agentId: "default_agent" });
const session = await agent.createSession({ workspaceDir: process.cwd() });
for await (const output of session.run([userText("创建 hello.txt 并写入 hi")], {
approve: async () => "allow", // 逐个工具审批
})) {
if (isCompleteModelMessage(output) && output.payload.type === "text") {
console.log(output.payload.text);
}
}
```
## 仓库结构
pnpm monorepo(TypeScript,Node >= 24)。一次安装交付四层组件,共享同一数据目录(`~/.penguin/data`)与同一消息协议(OmniMessage):
| 目录 | 包名 | 职责 |
| --- | --- | --- |
| [`packages/core`](packages/core) | `@prismshadow/penguin-core` | SDK 与引擎:ReAct 循环、OmniMessage 协议、LLM/Environment 接口契约、Agent State、Trace |
| [`packages/cli`](packages/cli) | `@prismshadow/penguin-cli` | `penguin` 命令:REPL、单次运行、模型与 Vault 配置、服务启动 |
| [`packages/server`](packages/server) | `@prismshadow/penguin-server` | Web 服务端:HTTP API + SSE 流式、多用户认证、Project 授权、用量统计 |
| [`packages/web`](packages/web) | `@prismshadow/penguin-web` | Web App:多 Session 对话、Agent/技能/模型管理、Trace 观测、评估中心 |
| [`packages/skills`](packages/skills) | `@prismshadow/penguin-skills` | 内置技能库(Agent 创建、Benchmark 设计、评估、优化等) |
| [`packages/landing`](packages/landing) | — | 产品落地页(本仓库官网) |
| [`packages/docs`](packages/docs) | — | 文档站(双语,部署于 `/docs/` 路径) |
职责按事实来源划分:**SDK** 负责协议与执行(消息解析、运行循环、工具),**Server** 负责多用户运行时(认证、SSE 流式、定时任务),`~/.penguin/data` 下的**文件层**承载一切可编辑与被记录的状态(Prompt、Skill、密钥、Trace)。逐项对应表见[架构总览 → 职责划分](https://prism-shadow.github.io/penguin-harness/docs/architecture)。
## 文档
文档站覆盖使用与设计两个层面:[产品介绍](https://prism-shadow.github.io/penguin-harness/docs/) · [快速开始](https://prism-shadow.github.io/penguin-harness/docs/quickstart) · [架构总览](https://prism-shadow.github.io/penguin-harness/docs/architecture) · [OmniMessage 协议](https://prism-shadow.github.io/penguin-harness/docs/omni-message) · [接口契约](https://prism-shadow.github.io/penguin-harness/docs/interfaces) · [Agent 运行循环](https://prism-shadow.github.io/penguin-harness/docs/agent-loop) · [CLI 参考](https://prism-shadow.github.io/penguin-harness/docs/cli) · [Server API](https://prism-shadow.github.io/penguin-harness/docs/server-api) · [配置参考](https://prism-shadow.github.io/penguin-harness/docs/configuration)
每页文档都带「复制 Markdown」按钮,可直接粘贴进模型上下文。
## 本地开发
```bash
pnpm install
pnpm build # 先构建:core 的导出指向 dist/
pnpm typecheck
pnpm test
pnpm dev:server # 服务端 127.0.0.1:7364
pnpm dev:web # Web App(Vite)127.0.0.1:7365,/api 代理到服务端
pnpm dev:docs # 文档站(Vite)127.0.0.1:7367
BASE_PATH=/ pnpm build:site # 按 Pages 部署的方式组装 落地页 + 文档
```
开发态模型凭据可复制 `.env.example` 为 `.env` 填写。E2E 测试走真实模型(`pnpm test:e2e`,需要 `DEEPSEEK_API_KEY`)。
## 许可证
[Apache-2.0](LICENSE) © 2026 Prism Shadow
+153
View File
@@ -0,0 +1,153 @@
#!/bin/sh
# PenguinHarness one-line installer.
#
# curl -fsSL https://github.com/Prism-Shadow/penguin-harness/releases/latest/download/install.sh | sh
#
# Options:
# PENGUIN_VERSION=vX.Y.Z pin a version (same as --version vX.Y.Z); default is the latest Release
# PENGUIN_INSTALL_DIR=<dir> install dir; default ~/.penguin
# --universal install the universal package (no bundled Node runtime; needs system Node >= 24)
#
# The data dir (~/.penguin/data) sits under the install home but is never touched by reinstall/upgrade (which only replace bin/lib/web/node).
#
# Docs: https://prism-shadow.github.io/penguin-harness/docs/installation
set -eu
REPO="https://github.com/Prism-Shadow/penguin-harness"
VERSION="${PENGUIN_VERSION:-}"
INSTALL_DIR="${PENGUIN_INSTALL_DIR:-$HOME/.penguin}"
BIN_DIR="$HOME/.local/bin"
UNIVERSAL=0
fail() {
echo "error: $1" >&2
exit 1
}
# --- Parse args (also passable via curl | sh -s -- --universal) ---
while [ $# -gt 0 ]; do
case "$1" in
--version)
[ $# -ge 2 ] || fail "--version requires a value (e.g. --version v1.0.0)"
VERSION="$2"
shift 2
;;
--universal)
UNIVERSAL=1
shift
;;
*)
fail "unknown option: $1"
;;
esac
done
# --- Detect platform: Linux/Darwin x64/arm64; other platforms should use the universal package ---
ASSET="penguin-universal.tar.gz"
if [ "$UNIVERSAL" -eq 0 ]; then
case "$(uname -s)" in
Linux) os="linux" ;;
Darwin) os="darwin" ;;
*) fail "unsupported OS: $(uname -s). Install Node.js >= 24, then re-run with --universal." ;;
esac
case "$(uname -m)" in
x86_64) arch="x64" ;;
aarch64 | arm64) arch="arm64" ;;
*) fail "unsupported architecture: $(uname -m). Install Node.js >= 24, then re-run with --universal." ;;
esac
ASSET="penguin-$os-$arch.tar.gz"
fi
# --- Universal package precheck: system Node >= 24 (platform packages bundle the runtime, so exempt) ---
if [ "$UNIVERSAL" -eq 1 ]; then
command -v node >/dev/null 2>&1 \
|| fail "the universal package needs Node.js >= 24 on PATH (none found)."
node_version="$(node --version)" # e.g. v24.18.0
v="${node_version#v}"
major="${v%%.*}"
if [ "$major" -lt 24 ]; then
fail "the universal package needs Node.js >= 24, found $node_version."
fi
fi
# --- Download (latest Release by default; PENGUIN_VERSION pins a version) ---
if [ -n "$VERSION" ]; then
BASE_URL="$REPO/releases/download/$VERSION"
else
BASE_URL="$REPO/releases/latest/download"
fi
TMP="$(mktemp -d)"
trap 'rm -rf "$TMP"' EXIT
echo "Downloading $BASE_URL/$ASSET ..."
curl -fSL --progress-bar "$BASE_URL/$ASSET" -o "$TMP/$ASSET" \
|| fail "download failed. Check the version tag and your network, then retry."
# --- SHA256 verify: only when .sha256 exists (skip on 404); warn and skip if no checksum tool ---
if curl -fsSL "$BASE_URL/$ASSET.sha256" -o "$TMP/$ASSET.sha256" 2>/dev/null; then
if command -v sha256sum >/dev/null 2>&1; then
(cd "$TMP" && sha256sum -c "$ASSET.sha256" >/dev/null 2>&1) || fail "checksum mismatch for $ASSET."
echo "Checksum OK."
elif command -v shasum >/dev/null 2>&1; then
(cd "$TMP" && shasum -a 256 -c "$ASSET.sha256" >/dev/null 2>&1) || fail "checksum mismatch for $ASSET."
echo "Checksum OK."
else
echo "warning: sha256sum/shasum not found; skipping checksum verification." >&2
fi
else
echo "warning: checksum file not available; skipping verification." >&2
fi
# --- Extract and swap into place: first move the new dirs into a staging area inside the install
# dir (same filesystem as the final location, so any slow cross-device copy happens before the
# old install is touched), then swap fast (rm old + same-disk mv is a rename; tiny window).
# No stale files after upgrade; the universal package has no node/, so cleanup lets the wrapper
# fall back to system Node. The data dir (~/.penguin/data) is untouched. ---
tar -xzf "$TMP/$ASSET" -C "$TMP"
[ -d "$TMP/penguin" ] || fail "unexpected archive layout: top-level penguin/ missing."
mkdir -p "$INSTALL_DIR"
STAGING="$INSTALL_DIR/.staging.$$"
rm -rf "$STAGING"
mkdir -p "$STAGING"
trap 'rm -rf "$TMP" "$STAGING"' EXIT
for d in bin lib web node; do
if [ -e "$TMP/penguin/$d" ]; then
mv "$TMP/penguin/$d" "$STAGING/$d"
fi
done
[ -x "$STAGING/bin/penguin" ] || fail "unexpected archive layout: bin/penguin missing."
rm -rf "$INSTALL_DIR/bin" "$INSTALL_DIR/lib" "$INSTALL_DIR/web" "$INSTALL_DIR/node"
for d in bin lib web node; do
if [ -e "$STAGING/$d" ]; then
mv "$STAGING/$d" "$INSTALL_DIR/$d"
fi
done
rm -rf "$STAGING"
[ -x "$INSTALL_DIR/bin/penguin" ] || fail "install incomplete: $INSTALL_DIR/bin/penguin missing."
# --- Symlink into ~/.local/bin and check PATH ---
mkdir -p "$BIN_DIR"
ln -sf "$INSTALL_DIR/bin/penguin" "$BIN_DIR/penguin"
case ":$PATH:" in
*":$BIN_DIR:"*) ;;
*)
echo ""
echo "note: $BIN_DIR is not on your PATH. Add it to your shell profile:"
case "${SHELL:-}" in
*/zsh) echo " echo 'export PATH=\"\$HOME/.local/bin:\$PATH\"' >> ~/.zshrc && source ~/.zshrc" ;;
*/bash) echo " echo 'export PATH=\"\$HOME/.local/bin:\$PATH\"' >> ~/.bashrc && source ~/.bashrc" ;;
*/fish) echo " fish_add_path \$HOME/.local/bin" ;;
*) echo " export PATH=\"\$HOME/.local/bin:\$PATH\"" ;;
esac
;;
esac
# --- Finish: print version and getting-started tips ---
installed_version="$("$INSTALL_DIR/bin/penguin" --version 2>/dev/null || echo "unknown")"
echo ""
echo "PenguinHarness $installed_version installed to $INSTALL_DIR"
echo ""
echo "Get started:"
echo " penguin --help # all commands"
echo " penguin web # start the Web UI at http://127.0.0.1:7364 (initial login: admin / admin123)"
echo " penguin server # headless server (PORT / HOST to override)"
+34
View File
@@ -0,0 +1,34 @@
{
"name": "penguin-harness",
"version": "0.0.1",
"private": true,
"type": "module",
"description": "PenguinHarness — TypeScript AI Agent (SDK + CLI).",
"engines": {
"node": ">=24"
},
"scripts": {
"format": "prettier --write .",
"format:check": "prettier --check .",
"typecheck": "pnpm -r typecheck",
"test": "pnpm -r test",
"test:e2e": "pnpm --filter @prismshadow/penguin-core test:e2e",
"build": "pnpm -r build && pnpm link:cli",
"link:cli": "pnpm --dir packages/cli link --global || echo '[link:cli] 未配置 pnpm 全局目录,跳过 penguin 全局链接(先运行一次 pnpm setup)'",
"penguin": "tsx packages/cli/src/index.ts",
"dev:server": "pnpm --filter @prismshadow/penguin-skills --filter @prismshadow/penguin-core build && pnpm --filter @prismshadow/penguin-server dev",
"dev:web": "pnpm --filter @prismshadow/penguin-skills --filter @prismshadow/penguin-core build && pnpm --filter @prismshadow/penguin-web dev",
"dev:docs": "pnpm --filter @prismshadow/penguin-docs dev",
"build:site": "node scripts/build-site.mjs"
},
"devDependencies": {
"@prismshadow/penguin-cli": "workspace:*",
"@prismshadow/penguin-core": "workspace:*",
"@types/node": "^24.0.0",
"prettier": "^3.9.4",
"tsx": "^4.20.0",
"typescript": "^5.6.0",
"vitest": "^2.1.0"
},
"packageManager": "pnpm@10.26.2"
}
+39
View File
@@ -0,0 +1,39 @@
# @prismshadow/penguin-cli
The PenguinHarness command line. Installs the `penguin` command: an interactive REPL, a one-shot task runner, model / vault configuration, and the launcher for the Web service.
```bash
npm install -g @prismshadow/penguin-cli # requires Node >= 24
```
```bash
penguin web # start the Web service and open http://127.0.0.1:7364
penguin server # same service, headless
penguin config model add --model-id deepseek-v4-pro --api-key sk-... --set-default
penguin run -m "Create hello.txt containing Hello, Penguin" # one Task, then exit
penguin chat # REPL: /compact, /exit, Ctrl-C interrupts
penguin chat --resume # resume the latest session
```
Tool calls go through an approval gate — `--approve allow-all` (default) `| deny-all | read-only | always-ask`. Data lives under `~/.penguin/data` (`PENGUIN_HOME` or `--root` override); model credentials come from the Project config or provider env vars (e.g. `DEEPSEEK_API_KEY`).
Prefer a one-line install with a bundled Node runtime? See the [installation guide](https://prism-shadow.github.io/penguin-harness/docs/installation).
## Documentation
- [Quickstart](https://prism-shadow.github.io/penguin-harness/docs/quickstart)
- [CLI Reference](https://prism-shadow.github.io/penguin-harness/docs/cli)
- [Configuration Reference](https://prism-shadow.github.io/penguin-harness/docs/configuration)
## Development
```bash
pnpm penguin <args> # run from source (repo root, via tsx)
pnpm --filter @prismshadow/penguin-cli build # tsup → dist/index.js (the penguin bin)
pnpm --filter @prismshadow/penguin-cli typecheck
pnpm --filter @prismshadow/penguin-cli test
```
Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0
+48
View File
@@ -0,0 +1,48 @@
{
"name": "@prismshadow/penguin-cli",
"version": "0.0.1",
"type": "module",
"description": "PenguinHarness CLI: interactive REPL and single-task runner over @prismshadow/penguin-core.",
"license": "Apache-2.0",
"repository": {
"type": "git",
"url": "git+https://github.com/Prism-Shadow/penguin-harness.git",
"directory": "packages/cli"
},
"bin": {
"penguin": "./dist/index.js"
},
"engines": {
"node": ">=24"
},
"scripts": {
"typecheck": "tsc --noEmit -p tsconfig.json",
"test": "vitest run --passWithNoTests",
"build": "tsup",
"penguin": "tsx src/index.ts"
},
"dependencies": {
"@prismshadow/agenthub": "^0.3.3",
"@prismshadow/penguin-core": "workspace:*",
"@prismshadow/penguin-server": "workspace:*",
"@prismshadow/penguin-skills": "workspace:*",
"commander": "^13.0.0",
"dotenv": "^17.0.0",
"smol-toml": "^1.3.0",
"yaml": "^2.5.0"
},
"devDependencies": {
"@types/node": "^24.0.0",
"tsup": "^8.3.0",
"tsx": "^4.20.0",
"typescript": "^5.6.0",
"vitest": "^2.1.0"
},
"files": [
"dist",
"LICENSE"
],
"publishConfig": {
"access": "public"
}
}
+142
View File
@@ -0,0 +1,142 @@
/**
* CLI tool-call approval.
*
* The CLI consumes the output stream of `session.run()`; within a turn, the engine invokes the
* injected `approve` callback for each tool_call.
* Docs: /docs/cli § "Approval modes (--approve)"; /docs/tools § "Approval".
*/
import { createInterface } from "node:readline";
import type { ApprovalDecision, ApproveFn } from "@prismshadow/penguin-core";
import { defaultMessages } from "./i18n.js";
import type { Messages } from "./i18n.js";
/** Valid string values for the `--approve` option (includes the default allow-all, so scripts can specify it explicitly and get the default behavior). */
const APPROVE_MODES = ["allow-all", "deny-all", "read-only", "always-ask"] as const;
/**
* Approval mode (derived from APPROVE_MODES, the single source of truth):
* - `allow-all`: auto-approve every tool (default mode);
* - `deny-all`: auto-reject every tool;
* - `read-only`: auto-approve read-only tools (permission === "r"), defer the rest to a human;
* - `always-ask`: interactive approval for each call.
*/
export type ApprovalMode = (typeof APPROVE_MODES)[number];
/**
* Resolve the approval mode from the CLI: read `--approve`, default `allow-all`; print a
* message and exit if the value is invalid.
*/
export function resolveApprovalMode(approve: string | undefined, t: Messages): ApprovalMode {
if (approve === undefined) return "allow-all";
const v = approve.trim().toLowerCase();
if ((APPROVE_MODES as readonly string[]).includes(v)) {
return v as ApprovalMode;
}
process.stderr.write(`${t.approveModeInvalid(approve)}\n`);
process.exit(1);
}
/**
* Build the `approve` callback for a given permission mode. `toolPermission` looks up a tool's
* permission level; `interactivePrompt` is the actual Q&A used when deferring to a human (run
* uses a one-off prompt, chat uses a persistent readline). Rendering the approval result is not
* done here — `context_engine` emits the decision as an `approval_decision` event, rendered by
* the frontend (see render.ts).
*/
export function makeApprove(args: {
mode: ApprovalMode;
toolPermission: (name: string) => "r" | "rw" | undefined;
interactivePrompt: ApproveFn;
}): ApproveFn {
const { mode, toolPermission, interactivePrompt } = args;
return async (toolCall) => {
const name = toolCall.payload.name;
switch (mode) {
case "allow-all":
return "allow";
case "deny-all":
return "deny";
case "read-only":
// Auto-approve read-only tools; defer read-write/unknown tools to a human.
if (toolPermission(name) === "r") return "allow";
return interactivePrompt(toolCall);
case "always-ask":
default:
return interactivePrompt(toolCall);
}
};
}
export interface PromptApprovalOptions {
/** Message set; resolved from the env var by default. */
t?: Messages;
/** Stream to read the approval answer from; defaults to `process.stdin`. */
input?: NodeJS.ReadableStream;
/** Stream to print the approval prompt to; defaults to `process.stdout`. */
output?: NodeJS.WritableStream;
}
/**
* Callback that rejects the pending approval while `promptApproval` is waiting; `null`
* otherwise. `penguin run` calls `denyActivePrompt()` from a single global SIGINT handler:
* Ctrl-C during approval collapses to "deny" (consistent with chat), and only interrupts the
* whole turn at other times. SIGINT is registered in exactly one place (run); promptApproval no
* longer attaches its own listener.
*/
let activePromptDeny: (() => void) | null = null;
export function denyActivePrompt(): boolean {
if (!activePromptDeny) return false;
activePromptDeny();
return true;
}
/**
* One-off interactive approval Q&A (for non-persistent REPL scenarios like `run`). The pending
* tool call has already been streamed above. Input-stream EOF/close is treated as a deny;
* Ctrl-C while waiting is collapsed to a deny by the caller (run) via `denyActivePrompt`. The
* readline instance is closed after reading.
*/
export function promptApproval(opts: PromptApprovalOptions = {}): Promise<ApprovalDecision> {
const input = opts.input ?? process.stdin;
const output = opts.output ?? process.stdout;
const t = opts.t ?? defaultMessages();
const rl = createInterface({ input, output });
return new Promise<ApprovalDecision>((resolve) => {
let resolved = false;
const finish = (decision: ApprovalDecision) => {
if (resolved) return;
resolved = true;
// Only clear our own slot: even under concurrent prompts (upstream already serializes
// this; this is a defensive check), don't clobber someone else's deny hook.
if (activePromptDeny === deny) activePromptDeny = null;
rl.close();
resolve(decision);
};
const deny = () => finish("deny");
// Ctrl-C during approval is turned into a "deny" via run's global SIGINT calling
// denyActivePrompt (no duplicate SIGINT listener registered here); input-stream EOF/close
// is likewise treated as a deny, to avoid hanging.
activePromptDeny = deny;
rl.on("close", () => finish("deny"));
rl.question(t.approvePrompt(), (answer) => {
// Tool approval defaults to allow: a bare Enter (empty input) counts as allow.
finish(parseApprovalAnswer(answer, "allow"));
});
});
}
/**
* Parse an approval/confirmation answer (trimmed, case-insensitive): `y`/`yes` → allow,
* `n`/`no` → deny, everything else (including empty input/bare Enter) → `fallback`. Tool
* approval defaults to allow (pass `"allow"`); exit/restart-style confirmations default to no
* (the default `"deny"`).
*/
export function parseApprovalAnswer(
answer: string,
fallback: ApprovalDecision = "deny",
): ApprovalDecision {
const normalized = answer.trim().toLowerCase();
if (normalized === "y" || normalized === "yes") return "allow";
if (normalized === "n" || normalized === "no") return "deny";
return fallback;
}
+369
View File
@@ -0,0 +1,369 @@
/**
* `penguin chat` — interactive REPL.
*
* penguin chat [--model-id <id>] [--provider <group>] [--project-id <id>] [--agent-id <id>]
* [--workspace <path>] [--approve <allow-all|deny-all|read-only|always-ask>]
*
* Each line of input starts one conversation turn; `/compact` proactively compacts the
* context (reason=manual); `/exit` or `/quit` exits.
* Uses the current directory when no Workspace is specified.
*
* Multi-line input: trailing `\` continues the line; when the terminal supports bracketed
* paste, a multi-line paste is treated as a single message (sent on Enter).
*
* Ctrl-C behavior (state-dependent): buffer has content -> clear it;
* awaiting approval -> deny; running -> abort the current turn and return to input;
* empty buffer -> show a y/N exit confirmation.
*
* Implementation notes: on a TTY, stdin is put into raw mode with bracketed paste enabled;
* stdin is piped through PasteFilter into a readline created with `terminal: true` — Ctrl-C
* is captured in-process by readline as 'SIGINT' (it never escapes as an OS signal killing
* the process group), and pasted content is held whole by PasteFilter (not split into
* multiple submits by embedded newlines).
* Docs: /docs/cli § "penguin chat".
*/
import { createInterface, type Interface } from "node:readline";
import type { Command } from "commander";
import { createAgent, userText } from "@prismshadow/penguin-core";
import type { ApprovalDecision, OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
import { StreamRenderer, dim, renderHistory } from "../render.js";
import { runTask } from "../task-loop.js";
import { parseApprovalAnswer, resolveApprovalMode } from "../approval.js";
import { LineComposer, PasteFilter } from "../input.js";
import type { Messages } from "../i18n.js";
export type ChatState = "idle" | "running" | "approving" | "confirming-exit";
export type SigintAction = "deny" | "abort" | "clear" | "confirm-exit" | "exit";
/** Pure decision: current state + whether the input buffer is non-empty -> the action Ctrl-C should perform. */
export function decideSigint(state: ChatState, hasBufferedInput: boolean): SigintAction {
if (state === "approving") return "deny";
if (state === "running") return "abort";
if (state === "confirming-exit") return "exit";
return hasBufferedInput ? "clear" : "confirm-exit";
}
interface RlInternals {
line: string;
cursor: number;
_refreshLine?: () => void;
}
const MAIN_PROMPT = "> ";
const CONT_PROMPT = "… ";
export function registerChatCommand(program: Command, t: Messages): void {
program
.command("chat")
.description(t.chat.desc)
.option("--model-id <id>", t.common.modelId)
.option("--provider <group>", t.common.provider)
.option("--project-id <id>", t.common.projectId)
.option("--agent-id <id>", t.common.agentId)
.option("--workspace <path>", t.common.workspace)
.option("--approve <mode>", t.common.approve)
.option("--resume [sessionId]", t.chat.resume)
.action(async (opts) => {
const mode = resolveApprovalMode(opts.approve, t);
const out = process.stdout;
const agent = await createAgent({
...(opts.agentId ? { agentId: opts.agentId } : {}),
...(opts.projectId ? { projectId: opts.projectId } : {}),
});
// --resume: resumes an existing Session. Workspace and
// Model follow the original Session and cannot be overridden; when omitted, resumes
// the current Agent's most recent Session.
let session;
if (opts.resume !== undefined) {
if (opts.workspace || opts.modelId || opts.provider) {
out.write(`${t.error(t.resumeNoOverride())}\n`);
process.exitCode = 1;
return;
}
const sessionId =
typeof opts.resume === "string" ? opts.resume : await agent.latestSessionId();
if (!sessionId) {
out.write(`${t.error(t.resumeNoSession())}\n`);
process.exitCode = 1;
return;
}
session = await agent.resumeSession({ sessionId });
} else {
session = await agent.createSession({
workspaceDir: opts.workspace ?? process.cwd(),
...(opts.modelId ? { modelId: opts.modelId } : {}),
...(opts.provider ? { provider: opts.provider } : {}),
});
}
const renderer = new StreamRenderer(out, t);
out.write(
`${t.header("chat", agent.state.agentId, session.workspaceDir, session.modelId)}\n` +
`${t.chatHints()}\n`,
);
// On resume, first render the history messages of the current context per Trace
// (full messages, including interrupted turns and their markers), then proceed to
// regular input.
if (session.resumedHistory) {
out.write(`${t.resumedBanner(session.sessionId, session.resumedHistory.length)}\n`);
renderHistory(session.resumedHistory, out);
}
// TTY: raw mode + bracketed paste + PasteFilter; non-TTY (pipe/test): read stdin directly.
const isTTY = Boolean(process.stdin.isTTY);
let pasteFilter: PasteFilter | null = null;
let inputStream: NodeJS.ReadableStream = process.stdin;
if (isTTY) {
process.stdin.setRawMode(true);
out.write("\x1b[?2004h");
pasteFilter = new PasteFilter();
process.stdin.pipe(pasteFilter);
inputStream = pasteFilter;
}
const rl = createInterface({
input: inputStream,
output: out,
terminal: isTTY,
});
const rli = rl as unknown as RlInternals;
const composer = new LineComposer();
let state: ChatState = "idle";
let closed = false;
let taskAbort: AbortController | null = null;
let pendingLine: ((line: string | null) => void) | null = null;
let pendingApproval: ((decision: ApprovalDecision) => void) | null = null;
const cleanup = () => {
if (!isTTY) return;
try {
out.write("\x1b[?2004l");
} catch {
/* ignore */
}
try {
process.stdin.setRawMode(false);
} catch {
/* ignore */
}
try {
if (pasteFilter) process.stdin.unpipe(pasteFilter);
} catch {
/* ignore */
}
try {
process.stdin.pause();
} catch {
/* ignore */
}
};
process.once("exit", cleanup);
if (pasteFilter) {
pasteFilter.on("paste", (text: string) => {
if (state !== "idle") return; // ignore paste while running
const { lineCount, normalized } = composer.pushPaste(text);
if (lineCount === 0) return;
out.write(`${normalized}\n`);
rl.setPrompt(CONT_PROMPT);
rl.prompt();
});
}
rl.on("line", (line) => {
if (state === "confirming-exit") {
if (parseApprovalAnswer(line) === "allow") {
rl.close();
} else {
state = "idle";
composer.reset();
out.write("\n");
rl.setPrompt(MAIN_PROMPT);
rl.prompt();
}
return;
}
if (state === "idle" && pendingLine) {
const { message } = composer.pushTypedLine(line);
if (message === undefined) {
// Continuation: show the continuation prompt and keep waiting.
rl.setPrompt(CONT_PROMPT);
rl.prompt();
} else {
const resolve = pendingLine;
pendingLine = null;
resolve(message);
}
return;
}
if (state === "approving" && pendingApproval) {
const resolve = pendingApproval;
pendingApproval = null;
// Tool approval defaults to allow: pressing Enter (empty input) is treated as allow.
resolve(parseApprovalAnswer(line, "allow"));
}
// running: ignore any line typed at this moment.
});
rl.on("SIGINT", () => {
const hasBuffer = rli.line.length > 0 || composer.hasPending();
const action = decideSigint(state, hasBuffer);
if (action === "deny") {
if (pendingApproval) {
const resolve = pendingApproval;
pendingApproval = null;
out.write("\n");
resolve("deny");
}
} else if (action === "abort") {
if (taskAbort && !taskAbort.signal.aborted) {
out.write(`\n${t.taskInterrupted()}\n`);
taskAbort.abort();
}
} else if (action === "clear") {
composer.reset();
rl.setPrompt(MAIN_PROMPT);
clearCurrentLine(rl, rli, out);
} else if (action === "confirm-exit") {
state = "confirming-exit";
rli.line = "";
rli.cursor = 0;
out.write("\n");
rl.setPrompt(t.confirmExit());
rl.prompt();
} else {
out.write("\n");
rl.close();
}
});
rl.on("close", () => {
closed = true;
if (pendingLine) {
const resolve = pendingLine;
pendingLine = null;
resolve(null);
}
});
const askLine = (): Promise<string | null> =>
new Promise((resolve) => {
if (closed) {
resolve(null);
return;
}
state = "idle";
pendingLine = resolve;
composer.reset();
rli.line = "";
rli.cursor = 0;
out.write("\n");
rl.setPrompt(MAIN_PROMPT);
rl.prompt();
});
// Interactive approval prompt: reuses the persistent readline, prompt text is
// localized; the tool call is already rendered above via streaming, so it is not
// re-rendered here.
const interactivePrompt = (_tc: OmniMessage<ToolCallPayload>): Promise<ApprovalDecision> =>
new Promise((resolve) => {
state = "approving";
pendingApproval = (decision) => {
state = "running";
resolve(decision);
};
rl.setPrompt(t.approvePrompt());
rl.prompt();
});
// Whether this Session already has a resumable Trace record: a resumed Session
// naturally has one; a new Session gets one starting from its first Task / compact
// (session_meta is written along with it). This decides whether to print the resume
// command example on exit.
let resumable = opts.resume !== undefined;
try {
for (;;) {
const line = await askLine();
if (line === null) break;
const text = line.trim();
if (text === "/exit" || text === "/quit") break;
if (text.length === 0) continue;
state = "running";
taskAbort = new AbortController();
try {
if (text === "/compact") {
// Proactive context compaction (Task boundary, reason=manual): the renderer
// prints compaction progress; Ctrl-C aborts the compaction via signal
// (preserving the original context). When there's nothing to compact (session
// just started / two consecutive /compact calls), the engine silently returns
// and we add one line of feedback here. Afterwards, settle the renderer's
// counters (endCompact) — compaction usage is already shown on the completion
// line and must not be counted again toward the next task's stats delta.
const startedAt = Date.now();
let sawMessage = false;
try {
for await (const msg of session.compact({
signal: taskAbort.signal,
})) {
sawMessage = true;
resumable = true;
renderer.handle(msg);
}
} finally {
renderer.endCompact(Date.now() - startedAt);
}
if (!sawMessage) out.write(`${t.compactNothing()}\n`);
} else {
resumable = true;
await runTask(session, [userText(text)], {
mode,
signal: taskAbort.signal,
renderer,
interactivePrompt,
t,
});
}
} catch (err) {
out.write(`\n${t.error(err instanceof Error ? err.message : String(err))}\n`);
} finally {
taskAbort = null;
state = "idle";
}
}
} finally {
rl.close();
cleanup();
session.dispose(); // tear down managed long-running command sessions to avoid leaking background processes
process.removeListener("exit", cleanup);
// On exit, print a dimmed resume command example: includes this
// session's Project / Agent options so the command can be copy-pasted directly;
// skipped when the Session has no Trace record yet (nothing to resume).
if (resumable) {
const command =
`penguin chat --resume ${session.sessionId}` +
(opts.projectId ? ` --project-id ${opts.projectId}` : "") +
(opts.agentId ? ` --agent-id ${opts.agentId}` : "");
out.write(`${dim(t.resumeHint(command))}\n`);
}
}
});
}
/** Clear the current input line and redraw the prompt (Ctrl-C clears the buffer when it has content). */
function clearCurrentLine(rl: Interface, rli: RlInternals, out: NodeJS.WritableStream): void {
rli.line = "";
rli.cursor = 0;
if (typeof rli._refreshLine === "function") {
rli._refreshLine();
} else {
out.write("\r\x1b[K");
rl.prompt(true);
}
}
+368
View File
@@ -0,0 +1,368 @@
/**
* `penguin config` — manages a Project's model credentials, default model, model list,
* Agent-level vault environment variables, and UI language.
*
* penguin config model add --model-id <upstream id> [--provider <group>] [--api-key <key>] [--context-window <n>] [--set-default] [--root <dir>]
* penguin config model default --model-id <upstream id> --provider <group> [--root <dir>]
* penguin config model vision --model-id <upstream id> --provider <group> [--root <dir>]
* penguin config model list [--root <dir>]
* penguin config vault set --key <name> --value <value> [--agent-id <id>] [--root <dir>]
* penguin config vault list [--agent-id <id>] [--root <dir>]
* penguin config vault remove --key <name> [--agent-id <id>] [--root <dir>]
* penguin config lang <en|zh>
*
* `--model-id` always takes the **upstream id** (the request id sent to AgentHub verbatim),
* which together with `--provider` forms a `(provider, model_id)` paired reference —
* **no string concatenation is ever performed**. For `model add`, --provider defaults to
* an inference from the built-in catalog (falling back to custom when inference fails);
* a new entry's client_type defaults according to the group's semantics (not set for
* first-party vendors; openai for custom / self-hosted groups / gateways, with the
* gateway's endpoint base URL pre-filled). For `model default` / `model vision`,
* --provider is **required**; core validation raises an error when the reference is not
* found in models. `--root` specifies the data root directory (priority: option >
* PENGUIN_HOME > ~/.penguin/data). The UI language is controlled by the PENGUIN_LANG
* environment variable; `config lang` writes it into the shell startup file and restarts
* the shell to take effect.
* Docs: /docs/cli § "penguin config".
*/
import { homedir } from "node:os";
import path from "node:path";
import { createInterface } from "node:readline";
import type { Command } from "commander";
import {
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
type ModelPricing,
type ModelRef,
type ProjectConfig,
addModel,
catalogEntryFor,
formatModelRef,
getModel,
inferProviderForUpstream,
loadAgentVault,
loadProjectConfig,
providerInfo,
removeVaultEntry,
resolveRoot,
setDefaultModel,
setVaultEntry,
setVisionModel,
} from "@prismshadow/penguin-core";
import { parseApprovalAnswer } from "../approval.js";
import { getMessages, maskApiKey, type Messages } from "../i18n.js";
import { applyLanguageToRc, restartShell } from "../lang-config.js";
/** Data root directory: the `--root` option takes priority (relative paths resolved against cwd), then PENGUIN_HOME / ~/.penguin/data. */
function resolveRootOption(root: string | undefined): string {
return root !== undefined ? path.resolve(root) : resolveRoot();
}
/**
* Renders the model list as column-aligned lines (the default model is marked with `*`;
* fully empty columns are omitted automatically). `provider` and `model_id` each occupy
* their own column (stored fields, never split apart); `vision` reflects the effective
* semantics (the TOML `vision` annotation takes priority, falling back to the catalog
* annotation — matched by the (provider, model_id) pair — and recorded as Y under
* "default = supported" when neither is present). Exported for unit tests.
*/
export function formatModelRows(cfg: ProjectConfig): string[] {
const cells = cfg.models.map((entry) => {
const cat = catalogEntryFor(entry.provider, entry.model_id);
const vision = entry.vision ?? cat?.supportsVision ?? true;
const isDefault =
cfg.default_model?.provider === entry.provider &&
cfg.default_model?.model_id === entry.model_id;
return {
provider: `${isDefault ? "* " : " "}${entry.provider}`,
model: entry.model_id,
vision: `vision=${vision ? "Y" : "-"}`,
context_window:
entry.context_window !== undefined ? `context_window=${entry.context_window}` : "",
client_type: entry.client_type ? `client_type=${entry.client_type}` : "",
pricing: entry.pricing
? `price=${entry.pricing.cache_read}/${entry.pricing.cache_write}/${entry.pricing.output}`
: "",
api_key: `api_key=${maskApiKey(entry.api_key)}`,
base_url: entry.base_url ? `base_url=${entry.base_url}` : "",
};
});
const columns = [
"provider",
"model",
"vision",
"context_window",
"client_type",
"pricing",
"api_key",
"base_url",
] as const;
const widths = columns.map((c) => Math.max(...cells.map((cell) => cell[c].length)));
const active = columns
.map((c, i) => ({ key: c, width: widths[i]! }))
.filter((col) => col.width > 0);
return cells.map((cell) =>
active
.map((col, i) => (i === active.length - 1 ? cell[col.key] : cell[col.key].padEnd(col.width)))
.join(" ")
.trimEnd(),
);
}
export function registerConfigCommand(program: Command, t: Messages): void {
const config = program.command("config").description(t.config.desc);
const model = config.command("model").description(t.config.modelDesc);
model
.command("add")
.description(t.config.addDesc)
.requiredOption("--model-id <id>", t.config.addModelId)
.option("--provider <group>", t.config.addProvider)
.option("--api-key <key>", t.config.addApiKey)
.option("--base-url <url>", t.config.addBaseUrl)
.option("--context-window <n>", t.config.addContextWindow, parseIntArg)
.option("--client-type <type>", t.config.addClientType)
// Tri-state: --vision marks it supported / --no-vision marks it unsupported / neither given keeps the existing value (defaults to supported).
.option("--vision", t.config.addVision)
.option("--no-vision", t.config.addNoVision)
.option("--price-cache-read <n>", t.config.addPriceCacheRead, parseFloatArg)
.option("--price-cache-write <n>", t.config.addPriceCacheWrite, parseFloatArg)
.option("--price-output <n>", t.config.addPriceOutput, parseFloatArg)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--set-default", t.config.addSetDefault, false)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
// --model-id takes the upstream id, paired with --provider as a reference
// (--provider defaults to catalog-based inference, falling back to custom); no
// concatenation is performed.
const modelId: string = opts.modelId;
const provider: string = opts.provider ?? inferProviderForUpstream(modelId);
const ref: ModelRef = { provider, model_id: modelId };
const before = await loadProjectConfig(root, opts.projectId);
const existed = getModel(before, ref) !== undefined;
// client_type default rule, only injected for new entries (updating an
// existing entry never overrides an explicit config): not set for first-party
// vendor groups (AgentHub auto-routes by upstream id, with env fallback keyed on
// id); defaults to openai for custom / self-hosted / gateway groups, with the
// gateway's endpoint base URL pre-filled as well.
const pInfo = providerInfo(provider);
const openAiDefault =
pInfo === undefined || pInfo.id === "custom" || pInfo.gatewayBaseUrl !== undefined;
const clientType: string | undefined =
opts.clientType ?? (!existed && openAiDefault ? "openai" : undefined);
const baseUrl: string | undefined =
opts.baseUrl ?? (!existed ? pInfo?.gatewayBaseUrl : undefined);
// Only collect explicitly given price fields, letting addModel merge them with the existing pricing per-field.
const pricing: Partial<ModelPricing> = {};
if (opts.priceCacheRead !== undefined) pricing.cache_read = opts.priceCacheRead;
if (opts.priceCacheWrite !== undefined) pricing.cache_write = opts.priceCacheWrite;
if (opts.priceOutput !== undefined) pricing.output = opts.priceOutput;
const cfg = await addModel(
root,
opts.projectId,
{
provider,
model_id: modelId,
...(opts.contextWindow !== undefined ? { context_window: opts.contextWindow } : {}),
...(clientType !== undefined ? { client_type: clientType } : {}),
...(opts.vision !== undefined ? { vision: opts.vision } : {}),
...(Object.keys(pricing).length > 0 ? { pricing } : {}),
...(opts.apiKey !== undefined ? { api_key: opts.apiKey } : {}),
...(baseUrl !== undefined ? { base_url: baseUrl } : {}),
},
{ setDefault: Boolean(opts.setDefault) },
);
const defaultRef = cfg.default_model && formatModelRef(cfg.default_model);
const line = existed
? t.modelUpdated(formatModelRef(ref), defaultRef)
: t.modelAdded(formatModelRef(ref), defaultRef);
process.stdout.write(`${line}\n`);
});
model
.command("default")
.description(t.config.defaultDesc)
.requiredOption("--model-id <id>", t.config.refModelId)
.requiredOption("--provider <group>", t.config.refProvider)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
// --model-id takes the upstream id, paired with the required --provider as a
// reference (no concatenation, no fuzzy matching); setDefaultModel raises an error
// when the reference is not found in models.
const ref: ModelRef = { provider: opts.provider, model_id: opts.modelId };
try {
await setDefaultModel(root, opts.projectId, ref);
} catch (err) {
process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`);
process.exitCode = 1;
return;
}
process.stdout.write(`${t.defaultModelSet(formatModelRef(ref))}\n`);
});
model
.command("vision")
.description(t.config.visionDesc)
.requiredOption("--model-id <id>", t.config.refModelId)
.requiredOption("--provider <group>", t.config.refProvider)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
// Paired reference semantics match `model default`; existence and vision=false semantics validation is handled by setVisionModel.
const ref: ModelRef = { provider: opts.provider, model_id: opts.modelId };
try {
await setVisionModel(root, opts.projectId, ref);
} catch (err) {
process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`);
process.exitCode = 1;
return;
}
process.stdout.write(`${t.visionModelSet(formatModelRef(ref))}\n`);
});
model
.command("list")
.description(t.config.listDesc)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
const cfg = await loadProjectConfig(root, opts.projectId);
if (cfg.models.length === 0) {
process.stdout.write(`${t.modelListEmpty()}\n`);
return;
}
process.stdout.write(`${t.modelListTitle()}\n`);
for (const line of formatModelRows(cfg)) {
process.stdout.write(`${line}\n`);
}
});
const vault = config.command("vault").description(t.config.vaultDesc);
vault
.command("set")
.description(t.config.vaultSetDesc)
.requiredOption("--key <name>", t.config.vaultKey)
.requiredOption("--value <value>", t.config.vaultValue)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--agent-id <id>", t.common.agentId, DEFAULT_AGENT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
try {
await setVaultEntry(root, opts.projectId, opts.agentId, opts.key, opts.value);
} catch (err) {
// Validation errors such as an invalid key name: print an explanation and exit with a non-zero code, without throwing a stack trace.
process.stderr.write(`${t.error(err instanceof Error ? err.message : String(err))}\n`);
process.exitCode = 1;
return;
}
process.stdout.write(`${t.vaultSet(opts.key)}\n`);
});
vault
.command("list")
.description(t.config.vaultListDesc)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--agent-id <id>", t.common.agentId, DEFAULT_AGENT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
const entries = Object.entries(await loadAgentVault(root, opts.projectId, opts.agentId));
if (entries.length === 0) {
process.stdout.write(`${t.vaultListEmpty()}\n`);
return;
}
process.stdout.write(`${t.vaultListTitle()}\n`);
const width = Math.max(...entries.map(([key]) => key.length));
for (const [key, value] of entries) {
process.stdout.write(`${key.padEnd(width)} ${maskApiKey(value)}\n`);
}
});
vault
.command("remove")
.description(t.config.vaultRemoveDesc)
.requiredOption("--key <name>", t.config.vaultKey)
.option("--project-id <id>", t.common.projectId, DEFAULT_PROJECT_ID)
.option("--agent-id <id>", t.common.agentId, DEFAULT_AGENT_ID)
.option("--root <dir>", t.common.root)
.action(async (opts) => {
const root = resolveRootOption(opts.root);
const vaultEntries = await loadAgentVault(root, opts.projectId, opts.agentId);
if (vaultEntries[opts.key] === undefined) {
process.stderr.write(`${t.vaultKeyMissing(opts.key)}\n`);
process.exitCode = 1;
return;
}
await removeVaultEntry(root, opts.projectId, opts.agentId, opts.key);
process.stdout.write(`${t.vaultRemoved(opts.key)}\n`);
});
config
.command("lang")
.description(t.config.langDesc)
.argument("<language>", t.config.langArg)
.action(async (language: string) => {
const lang = String(language).trim().toLowerCase();
if (lang !== "zh" && lang !== "en") {
process.stderr.write(`${t.langInvalid(String(language))}\n`);
process.exitCode = 1;
return;
}
const { rcPath } = await applyLanguageToRc(lang, {
shell: process.env.SHELL,
home: homedir(),
});
// The confirmation message is shown in the target language; the user must confirm before the shell restarts.
const m = getMessages(lang);
process.stdout.write(`${m.langSet(lang, rcPath)}\n`);
const interactive = Boolean(process.stdin.isTTY && process.stdout.isTTY);
if (interactive && (await confirmYes(m.langRestartConfirm()))) {
process.stdout.write(`${m.langRestart()}\n`);
restartShell(lang);
} else {
process.stdout.write(`${m.langRestartHint(rcPath)}\n`);
}
});
}
/** Interactive y/N confirmation; Ctrl-C (SIGINT) or input stream EOF/close are both treated as no, to avoid hanging. */
function confirmYes(prompt: string): Promise<boolean> {
const rl = createInterface({ input: process.stdin, output: process.stdout });
return new Promise<boolean>((resolve) => {
let done = false;
const finish = (value: boolean) => {
if (done) return;
done = true;
process.off("SIGINT", onSigint);
rl.close();
resolve(value);
};
const onSigint = () => finish(false);
process.once("SIGINT", onSigint);
rl.on("close", () => finish(false));
rl.question(prompt, (answer) => finish(parseApprovalAnswer(answer) === "allow"));
});
}
function parseIntArg(value: string): number {
const n = Number.parseInt(value, 10);
if (Number.isNaN(n)) {
throw new Error(`无效的整数:${value}`);
}
return n;
}
function parseFloatArg(value: string): number {
const n = Number.parseFloat(value);
if (Number.isNaN(n)) {
throw new Error(`无效的数值:${value}`);
}
return n;
}
+76
View File
@@ -0,0 +1,76 @@
/**
* `penguin run` — send a single Task in one shot.
*
* penguin run -m <msg> [--model-id <id>] [--provider <group>] [--workspace <path>]
* [--project-id <id>] [--agent-id <id>]
* [--approve <allow-all|deny-all|read-only|always-ask>]
*
* Uses the current directory when Workspace is unspecified; uses the Project's default model
* when model is unspecified. `--provider` is optional: when omitted, `--model-id` is resolved
* via resolveModelRef semantics (only matches when the exact value is globally unique in the
* config; ambiguity is an error). Defaults to interactive per-call approval; `--approve`
* selects the permission mode.
* Docs: /docs/cli § "penguin run".
*/
import type { Command } from "commander";
import { createAgent, userText } from "@prismshadow/penguin-core";
import { StreamRenderer } from "../render.js";
import { runTask } from "../task-loop.js";
import { denyActivePrompt, resolveApprovalMode } from "../approval.js";
import type { Messages } from "../i18n.js";
export function registerRunCommand(program: Command, t: Messages): void {
program
.command("run")
.description(t.run.desc)
.requiredOption("-m, --message <message>", t.run.message)
.option("--model-id <id>", t.common.modelId)
.option("--provider <group>", t.common.provider)
.option("--project-id <id>", t.common.projectId)
.option("--agent-id <id>", t.common.agentId)
.option("--workspace <path>", t.common.workspace)
.option("--approve <mode>", t.common.approve)
.action(async (opts) => {
const mode = resolveApprovalMode(opts.approve, t);
const agent = await createAgent({
...(opts.agentId ? { agentId: opts.agentId } : {}),
...(opts.projectId ? { projectId: opts.projectId } : {}),
});
const session = await agent.createSession({
workspaceDir: opts.workspace ?? process.cwd(),
...(opts.modelId ? { modelId: opts.modelId } : {}),
...(opts.provider ? { provider: opts.provider } : {}),
});
const out = process.stdout;
out.write(`${t.header("run", agent.state.agentId, session.workspaceDir, session.modelId)}\n`);
const controller = new AbortController();
const onSigint = () => {
// Single SIGINT handler: Ctrl-C during approval collapses to "deny this tool" (see
// approval.ts); at all other times it interrupts the whole turn.
if (denyActivePrompt()) return;
controller.abort();
};
process.on("SIGINT", onSigint);
const renderer = new StreamRenderer(out, t);
try {
const result = await runTask(session, [userText(opts.message)], {
mode,
signal: controller.signal,
renderer,
t,
});
// Task ended with an abort (LLM failure/reconnect exhausted/user interrupt): non-zero
// exit code, for scripts/CI to check.
if (result.aborted) process.exitCode = 1;
} finally {
process.off("SIGINT", onSigint);
session.dispose(); // Tear down managed long-running command sessions to avoid leaking background processes
}
out.write("\n");
});
}
+133
View File
@@ -0,0 +1,133 @@
/**
* `penguin server` / `penguin web` — starts the Web service.
*
* penguin server [--port <port>] [--host <host>]
* penguin web [--port <port>] [--host <host>] [--no-open]
*
* Both are entry points into the same service process: after setting PORT / HOST, it
* dynamically imports `@prismshadow/penguin-server` (whose entry point handles dotenv
* loading and graceful shutdown on its own), so the two never listen on separate ports
* in parallel. Port/host priority: command-line option > existing environment variable
* (including .env) > default 7364 / 127.0.0.1. `penguin web` additionally polls until the
* service is ready, prints the URL, and opens a browser per-platform (`--no-open`
* disables this).
* Docs: /docs/cli § "penguin server / penguin web".
*/
import { spawn } from "node:child_process";
import type { Command } from "commander";
import type { Messages } from "../i18n.js";
/** Default service port (deliberately avoids common defaults like 3000/8080). */
export const DEFAULT_PORT = 7364;
/** Default service listen host. */
export const DEFAULT_HOST = "127.0.0.1";
/**
* Resolves the listen port: command-line option takes priority, then the PORT
* environment variable, defaulting to 7364; throws if not an integer or out of the
* 0-65535 range. Exported for unit tests.
*/
export function resolvePort(option: string | undefined, env: string | undefined): number {
const raw = option ?? env;
if (raw === undefined || raw === "") return DEFAULT_PORT;
const port = Number(raw);
if (!Number.isInteger(port) || port < 0 || port > 65535) {
throw new Error(`Invalid port "${raw}". Use an integer between 0 and 65535.`);
}
return port;
}
/**
* Picks the command to open a browser per-platform. On win32, `start` treats the first
* quoted argument as the window title, so an extra empty title placeholder is passed.
* Exported for unit tests.
*/
export function browserCommand(platform: string, url: string): { command: string; args: string[] } {
if (platform === "darwin") return { command: "open", args: [url] };
if (platform === "win32") return { command: "cmd", args: ["/c", "start", "", url] };
return { command: "xdg-open", args: [url] };
}
/** URL used for the readiness probe and browser access: when listening on a wildcard address (0.0.0.0 / ::), access via 127.0.0.1 instead. Exported for unit tests. */
export function browserUrl(host: string, port: number): string {
const target = host === "0.0.0.0" || host === "::" ? "127.0.0.1" : host;
return `http://${target}:${port}/`;
}
/**
* Sets PORT / HOST then starts the service: the server entry point only reads
* process.env, and its dotenv loading never overrides existing environment variables,
* so the values written here are the ones that take effect (options take priority over
* .env and any pre-existing env vars).
*/
async function startServer(opts: {
port?: string;
host?: string;
}): Promise<{ host: string; port: number }> {
const port = resolvePort(opts.port, process.env.PORT);
const host = opts.host ?? process.env.HOST ?? DEFAULT_HOST;
process.env.PORT = String(port);
process.env.HOST = host;
await import("@prismshadow/penguin-server");
return { host, port };
}
/** Polls the service root path until it responds (any HTTP response counts as ready); keeps waiting on connection failure, returns false on timeout. */
async function waitForReady(url: string, timeoutMs = 15_000, intervalMs = 300): Promise<boolean> {
const deadline = Date.now() + timeoutMs;
for (;;) {
try {
// Each probe is capped at 1s: if the port is held by a non-HTTP program, the
// connection can succeed while the response hangs forever; without a timeout this
// would block the whole polling loop (the deadline check below would never run).
const res = await fetch(url, { signal: AbortSignal.timeout(1000) });
void res.body?.cancel();
return true;
} catch {
// The service isn't listening yet (or this probe timed out): keep polling.
}
if (Date.now() >= deadline) return false;
await new Promise((resolve) => setTimeout(resolve, intervalMs));
}
}
/** Opens the browser: spawn detached with output ignored; any failure is silently swallowed (failing to open doesn't affect the running service). */
function openBrowser(url: string): void {
const { command, args } = browserCommand(process.platform, url);
try {
const child = spawn(command, args, { detached: true, stdio: "ignore" });
child.on("error", () => {});
child.unref();
} catch {
// e.g. the browser command doesn't exist: ignore, the user can open it manually.
}
}
export function registerServeCommands(program: Command, t: Messages): void {
program
.command("server")
.description(t.serve.serverDesc)
.option("--port <port>", t.serve.port)
.option("--host <host>", t.serve.host)
.action(async (opts: { port?: string; host?: string }) => {
await startServer(opts);
});
program
.command("web")
.description(t.serve.webDesc)
.option("--port <port>", t.serve.port)
.option("--host <host>", t.serve.host)
.option("--no-open", t.serve.noOpen)
.action(async (opts: { port?: string; host?: string; open: boolean }) => {
const { host, port } = await startServer(opts);
const url = browserUrl(host, port);
const ready = await waitForReady(url);
if (!ready) {
process.stdout.write(`${t.webTimeout(url)}\n`);
return;
}
process.stdout.write(`${t.webReady(url)}\n`);
if (opts.open) openBrowser(url);
});
}
+390
View File
@@ -0,0 +1,390 @@
/**
* CLI text internationalization (i18n).
*
* Language comes from the `PENGUIN_LANG` env var (`en` / `zh`), defaulting to English (en) —
* independent of Project config or CLI options. This module centralizes all user-visible text:
* command/option help descriptions and runtime output, one implementation per language.
*/
/** UI language. */
export type Language = "en" | "zh";
/** Resolve the language from the env var; `zh` matches exactly, everything else falls back to English (see comment #2). */
export function resolveLanguage(): Language {
const v = (process.env.PENGUIN_LANG ?? "").trim().toLowerCase();
return v === "zh" ? "zh" : "en";
}
export interface Messages {
// —— Command/option help descriptions ——
cliDescription: string;
versionDesc: string;
common: {
projectId: string;
agentId: string;
modelId: string;
/** run/chat's --provider: pairs with --model-id; when omitted, resolved by unique match (ambiguity is an error). */
provider: string;
/** Data root directory option (priority: --root > PENGUIN_HOME > ~/.penguin/data). */
root: string;
workspace: string;
approve: string;
};
config: {
desc: string;
modelDesc: string;
addDesc: string;
addModelId: string;
addProvider: string;
addApiKey: string;
addBaseUrl: string;
addContextWindow: string;
addClientType: string;
addVision: string;
addNoVision: string;
addPriceCacheRead: string;
addPriceCacheWrite: string;
addPriceOutput: string;
addSetDefault: string;
defaultDesc: string;
visionDesc: string;
/** `model default` / `model vision`'s --model-id: the upstream request id (pairs with --provider as a reference). */
refModelId: string;
/** `model default` / `model vision`'s --provider: the provider group of the referenced entry (required). */
refProvider: string;
listDesc: string;
langDesc: string;
langArg: string;
vaultDesc: string;
vaultSetDesc: string;
vaultListDesc: string;
vaultRemoveDesc: string;
vaultKey: string;
vaultValue: string;
};
run: { desc: string; message: string };
chat: { desc: string; resume: string };
serve: {
serverDesc: string;
webDesc: string;
port: string;
host: string;
noOpen: string;
};
// —— Runtime output ——
header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string;
chatHints(): string;
confirmExit(): string;
taskInterrupted(): string;
error(message: string): string;
/** Approval prompt text (the tool call is already streamed above and directly precedes this prompt, so no index and no re-rendering). */
approvePrompt(): string;
/**
* Stats shown at the end of each Task: Session cumulative values plus this task's delta —
* context window length, Token usage, elapsed time. Delta strings carry their own sign
* (contextDelta can be negative after context is compacted), e.g.
* `[stats] context 4k (+1k) · tokens 6k (+1.2k) · 5.1s (+2.3s)`.
*/
taskStats(s: {
context: string;
contextDelta: string;
tokens: string;
tokensDelta: string;
elapsed: string;
elapsedDelta: string;
}): string;
/** Abort event label (may include a reason). */
abortLabel(reason?: string): string;
/** request_end ended with timeout/malformed: the engine retries (reconnect) carrying already-produced content; attempt is the retry count. */
reconnectLabel(status: "timeout" | "malformed", attempt: number): string;
/** compaction start event: indicates compaction in progress (mode is summarize/discard, reason is context/turns/manual). */
compactionStart(mode: string, reason: string): string;
/**
* compaction stop event: the compaction result (status is completed/failed/aborted;
* completed varies its text by mode). tokens is Token usage (same convention as the stats
* line: total = Session cumulative, delta = consumed by this compaction, carrying its own
* sign); when present it is appended at the end of the line, e.g. ` · tokens 14k (+6k)`.
*/
compactionStop(mode: string, status: string, tokens?: { total: string; delta: string }): string;
/** Prompt shown when `/compact` has nothing to compact (session just started / two consecutive compactions). */
compactNothing(): string;
/** Prompt for an invalid --approve mode. */
approveModeInvalid(value: string): string;
/** Render label for an approval decision (frontend renders the approval_decision event; one label each for allow/deny). */
approvalDecision(decision: "allow" | "deny"): string;
/** --resume is mutually exclusive with --workspace/--model-id (neither can change once the Session is created). */
resumeNoOverride(): string;
/** --resume given without a session id, and the current Agent has no Session at all. */
resumeNoSession(): string;
/** One-line prompt shown after a successful resume, before rendering history. */
resumedBanner(sessionId: string, messageCount: number): string;
/** Example resume command shown when the REPL exits (dim print; only when this session has a resumable record). */
resumeHint(command: string): string;
langInvalid(value: string): string;
langSet(lang: string, rcPath: string): string;
langRestartConfirm(): string;
langRestart(): string;
langRestartHint(rcPath: string): string;
/** Result output for model add/default/vision: the argument is the already-formatted pair reference (formatModelRef). */
modelAdded(model: string, defaultModel: string | undefined): string;
modelUpdated(model: string, defaultModel: string | undefined): string;
defaultModelSet(model: string): string;
visionModelSet(model: string): string;
modelListTitle(): string;
modelListEmpty(): string;
vaultSet(key: string): string;
vaultRemoved(key: string): string;
vaultKeyMissing(key: string): string;
vaultListTitle(): string;
vaultListEmpty(): string;
/** URL prompt once the `penguin web` service is ready. */
webReady(url: string): string;
/** Manual-open prompt after the `penguin web` ready-poll times out (15s). */
webTimeout(url: string): string;
}
function header(kind: "chat" | "run", agentId: string, workspace: string, model: string): string {
return `PenguinHarness ${kind} — agent=${agentId} workspace=${workspace} model=${model}`;
}
const en: Messages = {
cliDescription: "PenguinHarness CLI",
versionDesc: "output the version number",
common: {
projectId: "Project id",
agentId: "Agent id",
modelId: "Model to use (upstream model id; defaults to the Project default model)",
provider:
"Provider of --model-id; when omitted, the model id must match exactly one configured entry (ambiguity is an error)",
root: "Data root directory (overrides PENGUIN_HOME and ~/.penguin/data)",
workspace: "Workspace directory; must already exist (defaults to the current directory)",
approve:
"Approval mode: allow-all (auto-approve, default), deny-all (auto-reject), read-only (auto-approve read-only tools, prompt for the rest), always-ask (prompt per tool)",
},
config: {
desc: "Manage Project configuration",
modelDesc: "Manage model credentials and the default model",
addDesc: "Add or update a model, optionally writing a credential",
addModelId: "Upstream model id sent to AgentHub as-is (e.g. claude-sonnet-4-6)",
addProvider:
"Provider group stored alongside model_id; inferred from the builtin catalog when omitted, else custom",
addApiKey: "API key, stored inline in the Project's hidden .project_config.toml",
addBaseUrl: "Custom base URL",
addContextWindow: "Context window size (tokens)",
addClientType: "AgentHub client type (e.g. openai); inferred from model id when omitted",
addVision: "Mark the model as supporting image input (vision)",
addNoVision: "Mark the model as NOT supporting image input; omit both to keep current",
addPriceCacheRead: "Price per 1M tokens: cache read (USD)",
addPriceCacheWrite: "Price per 1M tokens: cache write (USD)",
addPriceOutput: "Price per 1M tokens: output (USD)",
addSetDefault: "Also set as the Project default model",
defaultDesc: "Set the Project default model",
visionDesc: "Set the vision model used by read_image for non-vision session models",
refModelId: "Upstream model id; forms the (provider, model_id) pair reference with --provider",
refProvider: "Provider group of the referenced entry (see `penguin config model list`)",
listDesc: "List the Project's models (API keys hidden)",
langDesc:
"Set the interface language (en|zh); persists PENGUIN_LANG to your shell startup file",
langArg: "Language: en or zh",
vaultDesc: "Manage an Agent's vault (environment variables injected into its shell commands)",
vaultSetDesc: "Set a vault environment variable (added or overwritten)",
vaultListDesc: "List vault environment variables (values masked)",
vaultRemoveDesc: "Remove a vault environment variable",
vaultKey: "Variable name (letters, digits and underscores; must not start with a digit)",
vaultValue: "Variable value, written to the Agent's agent_state/.vault.toml",
},
run: { desc: "Run a single Task", message: "Prompt for this Task" },
chat: {
desc: "Open the interactive REPL",
resume:
"Resume an existing Session (defaults to the agent's most recent one); workspace and model follow the original Session",
},
serve: {
serverDesc: "Start the Web service (HTTP API and the built-in frontend, same process)",
webDesc: "Start the Web service and open the UI in a browser once it is ready",
port: "Listen port (falls back to the PORT env var, default 7364)",
host: "Listen address (falls back to the HOST env var, default 127.0.0.1)",
noOpen: "Do not open a browser automatically",
},
header,
chatHints: () =>
"Type a message to start a conversation; end a line with \\; /compact to compact the context; /exit to quit; and Ctrl-C interrupts the current conversation.",
confirmExit: () => "Exit penguin? [y/N] ",
taskInterrupted: () => "[current conversation interrupted]",
error: (message) => `[error] ${message}`,
approvePrompt: () => "? Approve this tool call? [Y/n] ",
taskStats: (s) =>
`[stats] context ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · ${s.elapsed} (${s.elapsedDelta})`,
abortLabel: (reason) => `[abort]${reason ? `: ${reason}` : ""}`,
reconnectLabel: (status, attempt) =>
`[retry] ${status === "timeout" ? "connection timed out" : "response incomplete or unparseable"}; sending retry #${attempt}…`,
compactionStart: (mode, reason) =>
mode === "discard"
? `[compaction] discarding context (${reason})…`
: `[compaction] summarizing context (${reason})…`,
compactionStop: (mode, status, tokens) =>
(status === "completed"
? mode === "discard"
? "[compaction] done; old context discarded"
: "[compaction] done; continuing with the summarized context"
: `[compaction] ${status}; keeping the current context`) +
(tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""),
compactNothing: () => "[compaction] nothing to compact yet",
approveModeInvalid: (value) =>
`Invalid approval mode "${value}". Use allow-all, deny-all, read-only, or always-ask.`,
approvalDecision: (decision) => (decision === "allow" ? "✓ [approved]" : "× [denied]"),
resumeNoOverride: () =>
"--resume does not accept --workspace, --model-id or --provider: they follow the original Session and cannot change.",
resumeNoSession: () => "No session to resume: this agent has no recorded sessions yet.",
resumedBanner: (sessionId, messageCount) =>
`[resumed] ${sessionId} · ${messageCount} message${messageCount === 1 ? "" : "s"} in the current context`,
resumeHint: (command) => `To continue this conversation: ${command}`,
langInvalid: (value) => `Invalid language "${value}". Use en or zh.`,
langSet: (lang, rcPath) => `Language set to ${lang}; wrote PENGUIN_LANG to ${rcPath}.`,
langRestartConfirm: () => "Open a new shell now to apply? [y/N] ",
langRestart: () => "Opening a new shell with the new language (type exit to return)…",
langRestartHint: (rcPath) => `Open a new terminal, or run: source ${rcPath}`,
modelAdded: (model, def) => `Added model ${model}. Default model: ${def ?? "(unset)"}`,
modelUpdated: (model, def) => `Updated model ${model}. Default model: ${def ?? "(unset)"}`,
defaultModelSet: (model) => `Default model set to ${model}.`,
visionModelSet: (model) => `Vision model set to ${model}.`,
modelListTitle: () => "Configured models:",
modelListEmpty: () => "No models configured yet. Add one with `penguin config model add`.",
vaultSet: (key) => `Saved vault entry ${key}.`,
vaultRemoved: (key) => `Removed vault entry ${key}.`,
vaultKeyMissing: (key) => `Vault entry ${key} does not exist.`,
vaultListTitle: () => "Vault environment variables (values masked):",
vaultListEmpty: () => "The vault is empty. Add one with `penguin config vault set`.",
webReady: (url) => `Web UI ready: ${url}`,
webTimeout: (url) => `Server is not responding yet; open ${url} manually once it is ready.`,
};
const zh: Messages = {
cliDescription: "PenguinHarness CLI",
versionDesc: "输出版本号",
common: {
projectId: "Project id",
agentId: "Agent id",
modelId: "本次使用的模型(上游模型 id;默认 Project 默认模型)",
provider: "--model-id 的 provider 分组;省略时 model id 须在配置中精确唯一命中(歧义报错)",
root: "数据根目录(优先于 PENGUIN_HOME 与 ~/.penguin/data)",
workspace: "Workspace 目录,须为已存在目录(默认当前目录)",
approve:
"审批模式:allow-all(全部放行,缺省)、deny-all(全部拒绝)、read-only(自动放行只读工具,其余仍逐个询问)、always-ask(逐个询问)",
},
config: {
desc: "管理 Project 配置",
modelDesc: "管理模型 credential 与默认模型",
addDesc: "新增或更新一个模型,并可写入 credential",
addModelId: "上游模型 id(如 claude-sonnet-4-6,原样发给 AgentHub)",
addProvider: "与 model_id 分列存储的 provider 分组;缺省按内置目录推断,推断不出为 custom",
addApiKey: "API key,内联存入 Project 的隐藏文件 .project_config.toml",
addBaseUrl: "自定义 base url",
addContextWindow: "上下文窗口大小(token 数)",
addClientType: "AgentHub 客户端协议(如 openai);缺省由 model id 推断",
addVision: "标注该模型支持图片输入(视觉)",
addNoVision: "标注该模型不支持图片输入;两者都不给则保留原值",
addPriceCacheRead: "每百万 token 价格:缓存读取(USD)",
addPriceCacheWrite: "每百万 token 价格:缓存写入(USD)",
addPriceOutput: "每百万 token 价格:输出(USD)",
addSetDefault: "同时设为该 Project 的默认模型",
defaultDesc: "设置 Project 的默认模型",
visionDesc: "设置 read_image 代读用的视觉模型(供不支持图片的会话模型读图)",
refModelId: "上游模型 id;与 --provider 构成 (provider, model_id) 成对引用",
refProvider: "引用条目的 provider 分组(见 `penguin config model list`)",
listDesc: "列出当前 Project 的模型(API key 隐藏)",
langDesc: "设置界面语言(en|zh);将 PENGUIN_LANG 写入 shell 启动文件并持久化",
langArg: "语言:en 或 zh",
vaultDesc: "管理 Agent vault(注入该 Agent shell 命令的环境变量)",
vaultSetDesc: "写入一个 vault 环境变量(不存在则新增,存在则覆盖)",
vaultListDesc: "列出 vault 环境变量(值掩码显示)",
vaultRemoveDesc: "删除一个 vault 环境变量",
vaultKey: "变量名(字母、数字与下划线,不能以数字开头)",
vaultValue: "变量值,写入该 Agent 的 agent_state/.vault.toml",
},
run: { desc: "单次运行一个 Task", message: "本次 Task 的 Prompt" },
chat: {
desc: "打开交互式 REPL",
resume:
"恢复既有 Session 继续对话(缺省恢复当前 Agent 最近一次);Workspace 与模型沿用原 Session",
},
serve: {
serverDesc: "启动 Web 服务(HTTP API 与内置前端,同一进程)",
webDesc: "启动 Web 服务,就绪后用浏览器打开界面",
port: "监听端口(其次取环境变量 PORT,缺省 7364)",
host: "监听地址(其次取环境变量 HOST,缺省 127.0.0.1)",
noOpen: "不自动打开浏览器",
},
header,
chatHints: () =>
"输入消息发起对话;行尾 \\ 续行;/compact 压缩上下文;/exit 退出;Ctrl-C 中断对话。",
confirmExit: () => "确认退出 penguin?[y/N] ",
taskInterrupted: () => "[已中断当前对话]",
error: (message) => `[错误] ${message}`,
approvePrompt: () => "? 批准此工具调用?[Y/n] ",
taskStats: (s) =>
`[统计信息] 上下文 ${s.context} (${s.contextDelta}) · tokens ${s.tokens} (${s.tokensDelta}) · 用时 ${s.elapsed} (${s.elapsedDelta})`,
abortLabel: (reason) => `[已中断]${reason ? `:${reason}` : ""}`,
reconnectLabel: (status, attempt) =>
`[重试] ${status === "timeout" ? "连接超时或网络中断" : "响应不完整或无法解析"},正在发起第 ${attempt} 次重试……`,
compactionStart: (mode, reason) =>
mode === "discard"
? `[压缩] 正在丢弃旧上下文(${reason})……`
: `[压缩] 正在总结压缩上下文(${reason})……`,
compactionStop: (mode, status, tokens) =>
(status === "completed"
? mode === "discard"
? "[压缩] 完成,旧上下文已丢弃"
: "[压缩] 完成,已切换到摘要后的新上下文"
: `[压缩] ${status === "aborted" ? "已中断" : "失败"},保留当前上下文`) +
(tokens ? ` · tokens ${tokens.total} (${tokens.delta})` : ""),
compactNothing: () => "[压缩] 当前上下文为空,无需压缩",
approveModeInvalid: (value) =>
`无效的审批模式 "${value}"。请使用 allow-all、deny-all、read-only 或 always-ask。`,
approvalDecision: (decision) => (decision === "allow" ? "✓ [已批准]" : "× [已拒绝]"),
resumeNoOverride: () =>
"--resume 不接受 --workspace、--model-id 与 --provider:均沿用原 Session,创建后不可更换。",
resumeNoSession: () => "没有可恢复的 Session:当前 Agent 还没有任何会话记录。",
resumedBanner: (sessionId, messageCount) =>
`[已恢复] ${sessionId} · 当前上下文共 ${messageCount} 条消息`,
resumeHint: (command) => `继续本次对话:${command}`,
langInvalid: (value) => `无效的语言 "${value}"。请使用 en 或 zh。`,
langSet: (lang, rcPath) => `语言已设为 ${lang};已将 PENGUIN_LANG 写入 ${rcPath}。`,
langRestartConfirm: () => "现在打开新 shell 使其生效?[y/N] ",
langRestart: () => "正在打开使用新语言的新 shell(输入 exit 可返回)……",
langRestartHint: (rcPath) => `请打开新终端,或执行:source ${rcPath}`,
modelAdded: (model, def) => `已添加模型 ${model}。当前默认模型:${def ?? "(未设置)"}`,
modelUpdated: (model, def) => `已更新模型 ${model}。当前默认模型:${def ?? "(未设置)"}`,
defaultModelSet: (model) => `默认模型已设为 ${model}。`,
visionModelSet: (model) => `视觉模型已设为 ${model}。`,
modelListTitle: () => "已配置的模型:",
modelListEmpty: () => "尚未配置任何模型。用 `penguin config model add` 添加。",
vaultSet: (key) => `已保存 vault 条目 ${key}。`,
vaultRemoved: (key) => `已删除 vault 条目 ${key}。`,
vaultKeyMissing: (key) => `vault 条目 ${key} 不存在。`,
vaultListTitle: () => "vault 环境变量(值已掩码):",
vaultListEmpty: () => "vault 为空。用 `penguin config vault set` 添加。",
webReady: (url) => `Web 界面已就绪:${url}`,
webTimeout: (url) => `服务尚未就绪,请稍后手动打开 ${url}。`,
};
/** Get the message set for a language. */
export function getMessages(language: Language): Messages {
return language === "zh" ? zh : en;
}
/** Resolve the language from the env var and return its message set (the default used when no explicit `t` is given). */
export function defaultMessages(): Messages {
return getMessages(resolveLanguage());
}
/** Mask an API key: keep only a few trailing characters; return `-` when unconfigured. */
export function maskApiKey(apiKey: string | undefined): string {
if (!apiKey) return "-";
// Mask the whole thing when ≤12 chars: `****last4` reveals too much of a short secret (same threshold as the server-side mask).
if (apiKey.length <= 12) return "***";
return `****${apiKey.slice(-4)}`;
}
+46
View File
@@ -0,0 +1,46 @@
/**
* PenguinHarness CLI entry point.
*
* Only responsible for parsing CLI input into SDK arguments and rendering the streaming
* OmniMessage returned by the SDK.
* Loads .env on startup (e.g. locally configured ANTHROPIC_API_KEY / ANTHROPIC_BASE_URL).
*
* penguin config model add|default ...
* penguin chat ...
* penguin run --message ...
* penguin server|web ...
* Docs: packages/docs/content/cli.{zh,en}.md (site path /docs/cli).
*/
import "dotenv/config";
import { Command } from "commander";
import { VERSION } from "@prismshadow/penguin-core";
import { registerConfigCommand } from "./commands/config.js";
import { registerRunCommand } from "./commands/run.js";
import { registerChatCommand } from "./commands/chat.js";
import { registerServeCommands } from "./commands/serve.js";
import { defaultMessages } from "./i18n.js";
// Language comes from the PENGUIN_LANG env var (default en); used consistently for
// command/option descriptions and runtime output.
const t = defaultMessages();
const program = new Command();
program
.name("penguin")
.description(t.cliDescription)
.version(VERSION, "-v, --version", t.versionDesc);
registerConfigCommand(program, t);
registerRunCommand(program, t);
registerChatCommand(program, t);
registerServeCommands(program, t);
// Show help only when no subcommand is given (empty input); do not error.
program.action(() => {
program.outputHelp();
});
program.parseAsync(process.argv).catch((err: unknown) => {
process.stderr.write(`${err instanceof Error ? err.message : String(err)}\n`);
process.exitCode = 1;
});
+121
View File
@@ -0,0 +1,121 @@
/**
* CLI input-layer helpers: multi-line input and paste support.
*
* - `PasteFilter`: a Transform inserted between stdin and readline. Once terminal bracketed
* paste mode is enabled, pasted content is wrapped in `\x1b[200~` … `\x1b[201~`; this
* Transform strips that pair of markers, withholds the pasted content in between (not
* forwarded to readline, so internal newlines aren't split into multiple submissions), and
* emits it as a whole via a `paste` event. All other keystrokes are forwarded to readline
* unchanged, preserving line editing and Ctrl-C.
* - `LineComposer`: assembles "line-by-line input + paste blocks" into one complete message.
* A single trailing backslash `\` means line continuation; a paste block goes into the
* pending buffer as a whole and is sent on Enter.
*/
import { Transform, type TransformCallback } from "node:stream";
const PASTE_START = "\x1b[200~";
const PASTE_END = "\x1b[201~";
/**
* Return the trailing part of `data` that could be a prefix of `marker` (hold, kept for
* concatenation with the next chunk); the rest is ready to process immediately (emit). Handles
* the case where a marker straddles a data-chunk boundary.
*/
export function splitTrailingPartial(data: string, marker: string): { emit: string; hold: string } {
const max = Math.min(marker.length - 1, data.length);
for (let k = max; k > 0; k--) {
if (data.endsWith(marker.slice(0, k))) {
return { emit: data.slice(0, data.length - k), hold: data.slice(data.length - k) };
}
}
return { emit: data, hold: "" };
}
export class PasteFilter extends Transform {
private inPaste = false;
private pasteBuf = "";
private leftover = "";
override _transform(chunk: Buffer | string, _enc: BufferEncoding, cb: TransformCallback): void {
let data = this.leftover + chunk.toString("utf8");
this.leftover = "";
while (data.length > 0) {
if (!this.inPaste) {
const i = data.indexOf(PASTE_START);
if (i === -1) {
const { emit, hold } = splitTrailingPartial(data, PASTE_START);
if (emit) this.push(emit);
this.leftover = hold;
data = "";
} else {
if (i > 0) this.push(data.slice(0, i));
data = data.slice(i + PASTE_START.length);
this.inPaste = true;
this.pasteBuf = "";
}
} else {
const j = data.indexOf(PASTE_END);
if (j === -1) {
const { emit, hold } = splitTrailingPartial(data, PASTE_END);
this.pasteBuf += emit;
this.leftover = hold;
data = "";
} else {
this.pasteBuf += data.slice(0, j);
data = data.slice(j + PASTE_END.length);
this.inPaste = false;
const text = this.pasteBuf;
this.pasteBuf = "";
this.emit("paste", text);
}
}
}
cb();
}
}
/** Whether the line ends in a continuation (an odd number of trailing backslashes; an even count is treated as escaped literal backslashes). */
export function endsWithContinuation(line: string): boolean {
const trailing = line.match(/(\\+)$/)?.[1] ?? "";
return trailing.length % 2 === 1;
}
/**
* Assembles line-by-line input and paste blocks into a complete message.
* `pushTypedLine` returns `{ message }` when a message is ready, or `{}` while still
* continuing/pending.
*/
export class LineComposer {
private pending: string[] = [];
pushTypedLine(line: string): { message?: string } {
if (endsWithContinuation(line)) {
this.pending.push(line.slice(0, -1));
return {};
}
if (this.pending.length > 0) {
const lines = line === "" ? this.pending : [...this.pending, line];
this.pending = [];
return { message: lines.join("\n") };
}
return { message: line };
}
/** Accept a paste block (strip trailing blank lines, normalize newlines); it goes into the pending buffer as a whole, waiting to be sent on Enter. */
pushPaste(text: string): { lineCount: number; normalized: string } {
const norm = text.replace(/\r\n?/g, "\n").replace(/\n+$/, "");
if (norm.length === 0) return { lineCount: 0, normalized: "" };
const lines = norm.split("\n");
this.pending.push(...lines);
return { lineCount: lines.length, normalized: norm };
}
hasPending(): boolean {
return this.pending.length > 0;
}
reset(): void {
this.pending = [];
}
}
+102
View File
@@ -0,0 +1,102 @@
/**
* Language persistence: write `PENGUIN_LANG` into the user's shell startup file, then restart
* the shell so it takes effect.
*
* A child process can't modify its parent shell's environment variables directly, so
* `penguin config lang` uses a "write the startup file + restart the shell" approach: write
* `export PENGUIN_LANG=<lang>` into the shell startup file inside a marked block (idempotent,
* updates in place), then open an interactive shell carrying the new language env var. New
* terminals will read the variable from the startup file, so it persists.
*/
import { spawn } from "node:child_process";
import { mkdir, readFile, writeFile } from "node:fs/promises";
import { dirname, join } from "node:path";
import type { Language } from "./i18n.js";
const BEGIN = "# >>> PenguinHarness PENGUIN_LANG >>>";
const END = "# <<< PenguinHarness PENGUIN_LANG <<<";
export type ShellKind = "zsh" | "bash" | "fish" | "unknown";
export interface ShellRc {
kind: ShellKind;
/** Absolute path to the startup file. */
rcPath: string;
/** Generate the export line for a given language (shell-syntax specific). */
body(lang: Language): string;
}
/** Resolve the startup file and export syntax from `$SHELL`. Falls back to `~/.profile` for an unknown shell. */
export function resolveShellRc(shell: string | undefined, home: string): ShellRc {
const base = (shell ?? "").split("/").pop()?.toLowerCase() ?? "";
if (base.includes("fish")) {
return {
kind: "fish",
rcPath: join(home, ".config", "fish", "config.fish"),
body: (lang) => `set -gx PENGUIN_LANG ${lang}`,
};
}
if (base.includes("zsh")) {
return {
kind: "zsh",
rcPath: join(home, ".zshrc"),
body: (lang) => `export PENGUIN_LANG=${lang}`,
};
}
if (base.includes("bash")) {
return {
kind: "bash",
rcPath: join(home, ".bashrc"),
body: (lang) => `export PENGUIN_LANG=${lang}`,
};
}
return {
kind: "unknown",
rcPath: join(home, ".profile"),
body: (lang) => `export PENGUIN_LANG=${lang}`,
};
}
/** Insert or update the marked PenguinHarness block in place within the text; leaves the rest of the content unchanged. */
export function upsertBlock(content: string, bodyLine: string): string {
const block = `${BEGIN}\n${bodyLine}\n${END}`;
const begin = content.indexOf(BEGIN);
const end = content.indexOf(END);
if (begin !== -1 && end !== -1 && end > begin) {
const before = content.slice(0, begin);
const after = content.slice(end + END.length);
return `${before}${block}${after}`;
}
// Append at the end: leave a blank line before it if there's existing content.
if (content.length === 0) return `${block}\n`;
const sep = content.endsWith("\n") ? "" : "\n";
return `${content}${sep}\n${block}\n`;
}
/** Write the language into the shell startup file (creating the directory if needed). Returns the file path written and the shell kind. */
export async function applyLanguageToRc(
lang: Language,
opts: { shell: string | undefined; home: string },
): Promise<{ rcPath: string; kind: ShellKind }> {
const rc = resolveShellRc(opts.shell, opts.home);
await mkdir(dirname(rc.rcPath), { recursive: true });
let content = "";
try {
content = await readFile(rc.rcPath, "utf8");
} catch {
/* File doesn't exist yet; treat as empty content */
}
await writeFile(rc.rcPath, upsertBlock(content, rc.body(lang)), "utf8");
return { rcPath: rc.rcPath, kind: rc.kind };
}
/** Open an interactive shell carrying the new language env var; this process exits when the user exits that shell. */
export function restartShell(lang: Language): void {
const shell = process.env.SHELL || "/bin/zsh";
const child = spawn(shell, ["-i"], {
stdio: "inherit",
env: { ...process.env, PENGUIN_LANG: lang },
});
child.on("exit", (code) => process.exit(code ?? 0));
child.on("error", () => process.exit(1));
}
+874
View File
@@ -0,0 +1,874 @@
/**
* CLI streaming renderer.
*
* Rendering rule: **only the streaming `partial_*` variants of model_msg are rendered**;
* complete (non-streaming) model_msg is never rendered. A complete message's content has
* already been delivered by its corresponding `partial_*` stream, so re-rendering it would
* be redundant. `partial_*` is written out token by token as it arrives.
* event_msg is not message rendering and is handled separately: `token_usage` accumulates
* and is summarized in the `[stats]` line at task end, `approval_decision` prints one line
* with the approval result, `abort` prints one line noting the interruption, and each of
* `compaction_begin`/`compaction_end` prints one line of compaction progress;
* `session_meta` is never rendered.
*
* **Screen lock (concurrent tools)**: tools run concurrently and asynchronously, so
* messages may arrive interleaved. The renderer queues internally to guarantee:
* - a streaming segment (the LLM's text/thinking/tool_call stream, or a given tool's
* output stream start->delta->stop) holds the screen until stop, while other messages
* queue up;
* - all output is locked while waiting for user input (the approval prompt,
* `beginUserPrompt`/`endUserPrompt`);
* - when the head of the queue is held, the holder's own subsequent messages are let
* through first (preserving in-segment order), avoiding deadlock.
*
* **Pairing tags**: a tool call and its output may be separated by several segments, so
* both are tagged with a shared word for pairing: the call line reads
* `[tool-653] $ cmd`, the output line `[tool-653] >> ...` (653 being the last 3
* characters of tool_call_id); nested (subagent) tools use
* `[agent-f2a-tool-653] $ cmd` (f2a being the last 3 characters of the direct child
* Session id). Approval lines carry no tag (they immediately follow the matching call
* line, so context makes the pairing clear): `[approved]`.
*
* **Nested sub-session messages** (those carrying an origin) are handled separately:
* child tool calls (so the user can see what the subagent is calling before approval)
* and child approval results are rendered, and child token_usage counts toward this
* task's delta and the Session total; everything else (child text/thinking, etc.) is
* not rendered — the child Agent's final text is already streamed through the parent
* tool's output gutter.
*
* No third-party color library is used; only minimal ANSI escapes.
*/
import { isEventMessage, isModelMessage } from "@prismshadow/penguin-core";
import type {
AbortPayload,
ApprovalDecision,
ApprovalDecisionPayload,
CompactionBeginPayload,
CompactionEndPayload,
MessageOrigin,
OmniMessage,
PartialTextPayload,
PartialThinkingPayload,
PartialToolCallPayload,
PartialToolCallOutputPayload,
RequestEndPayload,
TokenUsagePayload,
ToolCallPayload,
} from "@prismshadow/penguin-core";
import { renderPartialToolCall } from "./tool-render.js";
import { defaultMessages } from "./i18n.js";
import type { Messages } from "./i18n.js";
const DIM = "\x1b[2m";
const CYAN = "\x1b[36m";
const RESET = "\x1b[0m";
export function dim(text: string): string {
return `${DIM}${text}${RESET}`;
}
/** Colors a tool call line cyan, distinguishing it from body text/thinking (review comment #5). */
function cyan(text: string): string {
return `${CYAN}${text}${RESET}`;
}
/** Takes the last 3 characters of an id as the on-screen pairing number. */
function shortId(id: string): string {
return id.slice(-3);
}
/**
* On-screen pairing tag for a tool call/output: main-session tools ->
* `tool-<last 3 chars of id>`; nested (subagent) tools ->
* `agent-<last 3 chars of direct child Session>-tool-<last 3 chars of id>`.
*/
function callTag(toolCallId: string, origin?: readonly MessageOrigin[]): string {
const tid = `tool-${shortId(toolCallId)}`;
return origin && origin.length > 0 ? `agent-${shortId(origin[origin.length - 1]!)}-${tid}` : tid;
}
/** Converts a token count to a human-readable abbreviation: 1234->1.2k, 1500000->1.5M, <1000 unchanged. */
export function humanizeTokens(n: number): string {
const abs = Math.abs(n);
if (abs < 1000) return `${n}`;
if (abs < 1_000_000) {
const v = n / 1000;
return `${trimZero(v)}k`;
}
const v = n / 1_000_000;
return `${trimZero(v)}M`;
}
/** Keeps one decimal place but drops a trailing `.0`. */
function trimZero(v: number): string {
const s = v.toFixed(1);
return s.endsWith(".0") ? s.slice(0, -2) : s;
}
/** Adds an explicit sign to a delta string: non-negative gets a `+` prefix, negative already has its own `-` (context can go negative after compaction shrinks it). */
function signedDelta(formatted: string): string {
return formatted.startsWith("-") ? formatted : `+${formatted}`;
}
/** Converts milliseconds into a human-readable duration: `820ms`, `2.3s`, `1m3s`. */
function humanizeDuration(ms: number): string {
if (ms < 1000) return `${Math.round(ms)}ms`;
const s = ms / 1000;
if (s < 60) return `${trimZero(s)}s`;
const m = Math.floor(s / 60);
return `${m}m${Math.round(s % 60)}s`;
}
export function formatAbort(p: AbortPayload, t: Messages): string {
return dim(t.abortLabel(p.reason ?? undefined));
}
/**
* Statically renders resumed history messages (`--resume`: full-message semantics, no
* partial_*, including interrupted messages and their markers). Uses the
* same color scheme as streaming rendering: user input `> `, dim thinking, cyan tool
* calls, dim tool-output gutter; a message whose `stop_reason` isn't completed gets a
* dim marker appended at the end of its line.
*/
export function renderHistory(
messages: OmniMessage[],
out: NodeJS.WritableStream,
t: Messages = defaultMessages(),
): void {
for (const msg of messages) {
if (isEventMessage(msg)) {
const p = msg.payload as { type?: string } & AbortPayload;
if (p.type === "abort") out.write(`${formatAbort(p, t)}\n`);
continue;
}
if (!isModelMessage(msg)) continue;
const p = msg.payload as {
type?: string;
role?: string;
text?: string;
thinking?: string;
name?: string;
arguments?: string;
output?: string;
images?: string[];
tool_call_id?: string;
stop_reason?: string;
};
const marker = p.stop_reason && p.stop_reason !== "completed" ? dim(` [${p.stop_reason}]`) : "";
switch (p.type) {
case "text":
if (p.role === "user") out.write(`\n> ${p.text ?? ""}\n`);
else out.write(`${p.text ?? ""}${marker}\n`);
break;
case "image_url":
out.write(`\n> ${dim("[image]")}\n`);
break;
case "thinking":
out.write(`${dim(p.thinking ?? "")}${marker}\n`);
break;
case "tool_call": {
const preview =
renderPartialToolCall(p.name ?? "", p.arguments ?? "") ?? `${p.name} ${p.arguments}`;
out.write(`${cyan(`[${callTag(p.tool_call_id ?? "")}] ${preview}`)}${marker}\n`);
break;
}
case "tool_call_output": {
const tag = callTag(p.tool_call_id ?? "");
for (const line of (p.output ?? "").split("\n")) {
out.write(`${DIM}[${tag}] >> ${RESET}${line}\n`);
}
// Attached images aren't rendered by the terminal; print one placeholder line per image.
for (const _ of p.images ?? []) {
out.write(`${DIM}[${tag}] >> [image]${RESET}\n`);
}
break;
}
default:
break; // inline_data / inline_thinking etc.: not shown in static history rendering for now
}
}
}
/**
* Streaming renderer: writes the OmniMessage stream to the output stream. The display
* text for tool calls is decided locally by `tool-render.ts`; it no longer accepts a
* tool-render callback from core (rendering has moved down into the CLI).
*/
export class StreamRenderer {
private readonly out: NodeJS.WritableStream;
private readonly t: Messages;
/** Pending render queue: while the screen is held (a streaming segment is in progress / awaiting user input), messages queue up here. */
private pending: OmniMessage[] = [];
/** The streaming segment currently holding the screen ("llm" or "out:<tool_call_id>"); null = idle. */
private holder: string | null = null;
/** Awaiting user input (approval prompt): locks the screen, all messages queue up. */
private promptActive = false;
/** Key of the call the current interactive prompt belongs to (the tool_call passed to beginUserPrompt); null = unattached. */
private promptKey: string | null = null;
/**
* Approval results for **other calls** that arrive during an interactive prompt
* (concurrent subagent / auto-approval paths): must not be written straight into the
* middle of an unanswered prompt, so they're deferred and rendered in order once
* endUserPrompt unlocks the screen.
*/
private deferredDecisions: Array<{
toolCall: OmniMessage<ToolCallPayload>;
decision: ApprovalDecision;
}> = [];
/** Reentrancy guard for drain. */
private draining = false;
/**
* Keys (origin chain + tool_call_id) of call lines already **rendered in place** from
* a complete message: rendered ahead of the streaming copy at approval time, so any
* streaming/nested copy that arrives afterward is deduplicated and skipped based on
* this set. Guarantees the approval prompt always immediately follows its matching
* call line (messages arrive through an async pipeline and may arrive later than the
* approval callback). Cleared at task end (see endTask).
*/
private ensuredCallLines = new Set<string>();
/** Call-line key of the last **content line actually written**; cleared once anything else is written. Used to check whether a call line is still adjacent to the current position. */
private lastLineKey: string | null = null;
/** Calls whose result has already been rendered in place at the approval callback (keyed the same as callLineKey); deduplicates a later-arriving approval_decision event. */
private renderedDecisions = new Set<string>();
/** Whether we're currently mid-way through a streaming line (text/thinking/tool output) that hasn't been newline-terminated yet. */
private inLine = false;
/** Whether we're currently in a dim span (thinking), used to know when to emit RESET. */
private inDim = false;
/** Whether tool-call output is at the start of a line (decides whether the gutter needs to be written). */
private toolOutLineStart = true;
/** Buffer for partial_tool_call; each delta streams out the newly appended suffix of the preview. */
private partialToolCalls = new Map<
string,
{ name: string; arguments: string; lastPreview: string }
>();
/** The partial_tool_call currently being rendered as a stream. */
private partialToolCallLineId: string | null = null;
/** This task's accumulated request tokens, the parent session's cumulative Session tokens, and whether this task has seen any usage. */
private taskTokens = 0;
private sessionTotal = 0;
private hasUsage = false;
/**
* Session-level accumulation of sub-session (subagent) request tokens: persists across
* tasks, never reset by endTask. The Token total shown to the user =
* sessionTotal + subagentTotal, using the same accounting as this task's delta
* (parent + child), guaranteeing the sum of per-task deltas never exceeds the
* cumulative increase.
*/
private subagentTotal = 0;
/** Current context (= input+output = total of the most recent request), the context at the end of the previous task, and cumulative Session elapsed time (ms). */
private contextNow = 0;
private contextAtTaskStart = 0;
private sessionElapsedMs = 0;
/**
* Compaction in progress (between a pair of parent-session compaction events): any
* parent-session token_usage arriving during this window is compaction-request usage —
* it does not update the context accounting (the actual usage after compaction is
* reported by the next normal request); it's accumulated into compactionTokens so the
* compaction-completion line can show "usage this time", and also staged into
* pendingCompactionTokens pending final attribution (see below).
*/
private compactionActive = false;
private compactionTokens = 0;
/**
* Staged compaction usage: when a compaction event arrives, it's not yet known whether
* it happened **mid-turn** (a normal request_end still follows in this turn ->
* attribute to this turn) or **after the turn ended** (nothing follows -> don't
* attribute to this turn). Mid-turn compaction is folded into taskTokens at the next
* non-compaction request_end; compaction after the turn ended is discarded when
* endTask/endCompact settles up. Uses the same accounting as the Web side
* (stream-model / task-stats).
*/
private pendingCompactionTokens = 0;
/**
* Timestamps (ms) of this task's first (non-session_meta) message and its last
* **non-compaction** request_end: the elapsed time shown in the stats line = the
* latter minus the former. A mid-turn compaction naturally falls within this span and
* is counted; one after the turn ends falls after it and is naturally excluded
* (consistent with "the last request_end before stats were queried"). The degenerate
* case of a turn with no request_end at all falls back to the externally supplied
* wall-clock elapsed time.
*/
private taskFirstTsMs: number | null = null;
private taskLastReqEndMs: number | null = null;
/** Terminal state (timeout/malformed) of the previous request: the next request_begin is a retry, at which point a notice is printed. */
private pendingRetry: "timeout" | "malformed" | null = null;
/** Number of retries already initiated (increments on consecutive failures, reset once a request completes normally). */
private reconnectRun = 0;
constructor(out: NodeJS.WritableStream = process.stdout, t: Messages = defaultMessages()) {
this.out = out;
this.t = t;
}
handle(msg: OmniMessage): void {
this.pending.push(msg);
this.drain();
}
/**
* Enters user interaction (approval prompt): first ensures the call line awaiting
* approval is **immediately adjacent to the current position** (if unrendered or
* separated by other output, render it in place from the complete message directly),
* then finishes the current line and locks the screen, queuing any messages that
* arrive in the meantime — guaranteeing "tool call -> approval prompt" stay adjacent,
* for both the main Agent and subagents.
*/
beginUserPrompt(toolCall?: OmniMessage<ToolCallPayload>): void {
if (toolCall) this.ensureAdjacentCallLine(toolCall);
this.finishLine();
this.promptActive = true;
this.promptKey = toolCall
? this.callLineKey(toolCall.payload.tool_call_id, toolCall.origin)
: null;
}
/**
* Renders one approval result, guaranteeing "tool call -> (approval prompt) ->
* approval result" appear consecutively:
* - interactive path: called **before** the prompt ends and unlocks (nothing else can
* preempt output while the lock is held);
* - auto-approval path (allow-all etc., no prompt): if the call line isn't adjacent,
* render it in place first, then write the result, so they appear as a pair.
* Idempotent (a given call's result is rendered only once); a subsequent
* approval_decision event arriving through the pipeline is deduplicated by key.
*/
noteApprovalDecision(toolCall: OmniMessage<ToolCallPayload>, decision: ApprovalDecision): void {
const key = this.callLineKey(toolCall.payload.tool_call_id, toolCall.origin);
// The screen is locked by **another call's** interactive prompt (e.g. auto-approval
// of a concurrent subagent): must not write straight into the middle of an
// unanswered prompt, so defer until unlocked; this prompt's own result still renders
// in place as usual (it holds the lock).
if (this.promptActive && this.promptKey !== key) {
this.deferredDecisions.push({ toolCall, decision });
return;
}
if (this.renderedDecisions.has(key)) return;
this.renderedDecisions.add(key);
this.ensureAdjacentCallLine(toolCall);
this.finishLine();
this.out.write(`${dim(this.t.approvalDecision(decision))}\n`);
this.lastLineKey = null;
}
/** Call-line dedup key: origin chain + tool_call_id (parent/child session ids may collide, so the chain is needed to disambiguate). */
private callLineKey(id: string, origin?: readonly MessageOrigin[]): string {
return `${origin?.join("/") ?? ""}:${id}`;
}
/**
* Ensures a given tool_call's call line is adjacent to the current position: if it
* isn't the last content line (unrendered, or separated by other output since), it is
* (re-)rendered in place from the complete message, and registered so any late
* streaming/nested copy is deduplicated and skipped.
*/
private ensureAdjacentCallLine(tc: OmniMessage<ToolCallPayload>): void {
const key = this.callLineKey(tc.payload.tool_call_id, tc.origin);
// The call line is already the last content line and its streaming segment has
// already finished: already adjacent, nothing to do. If it's still mid-stream (the
// line may show only half the arguments), re-render the full line in place and
// register it for dedup — otherwise a late tail delta arriving after unlock would
// start a duplicate call line, breaking the "call -> prompt -> result" adjacency
// invariant.
if (this.lastLineKey === key && this.partialToolCallLineId !== tc.payload.tool_call_id) {
return;
}
this.renderCallLine(tc.payload, tc.origin, key);
}
/** Renders one call line in place from a complete tool_call and registers its dedup key (shared by in-place approval rendering and nested rendering). */
private renderCallLine(
p: ToolCallPayload,
origin: readonly MessageOrigin[] | undefined,
key: string,
): void {
this.ensuredCallLines.add(key);
const preview = renderPartialToolCall(p.name, p.arguments) ?? `${p.name} ${p.arguments}`;
this.finishLine();
this.out.write(`${cyan(`[${callTag(p.tool_call_id, origin)}] ${preview}`)}\n`);
this.lastLineKey = key;
}
/** User interaction ends: unlocks the screen, first renders approval results deferred during the lock, then drains the queue. */
endUserPrompt(): void {
this.promptActive = false;
this.promptKey = null;
this.flushDeferredDecisions();
this.drain();
}
/** Renders approval results deferred during the interactive prompt (call line + result as a pair; called after unlocking). */
private flushDeferredDecisions(): void {
const deferred = this.deferredDecisions;
if (deferred.length === 0) return;
this.deferredDecisions = [];
for (const d of deferred) this.noteApprovalDecision(d.toolCall, d.decision);
}
/** Streaming segment ownership: the LLM stream (text/thinking/tool_call share one stream serially) or a given tool's output stream; null = atomic message. */
private streamOwner(msg: OmniMessage): string | null {
if (msg.origin && msg.origin.length > 0) return null; // nested messages render as atomic lines
if (!isModelMessage(msg)) return null;
const type = msg.payload.type;
if (type === "partial_text" || type === "partial_thinking" || type === "partial_tool_call") {
return "llm";
}
if (type === "partial_tool_call_output") {
return `out:${(msg.payload as PartialToolCallOutputPayload).tool_call_id}`;
}
return null;
}
private isStop(msg: OmniMessage): boolean {
return (msg.payload as { event_type?: string }).event_type === "stop";
}
/**
* Drains the pending render queue. The same streaming segment (start->delta->stop)
* holds the screen until stop, while other messages queue up; while the screen is
* held, the holder's own subsequent messages are let through first (preserving
* in-segment order, while other messages keep their arrival order); nothing is let
* through while awaiting user input.
*/
private drain(): void {
if (this.draining) return;
this.draining = true;
try {
while (!this.promptActive && this.pending.length > 0) {
if (this.holder === null) {
const msg = this.pending.shift()!;
const owner = this.streamOwner(msg);
if (owner !== null) this.holder = this.isStop(msg) ? null : owner;
this.renderNow(msg);
continue;
}
// Screen is held: let through all of the holder's own messages in a single
// pass (avoiding the quadratic cost of rescanning from the queue head after
// each message); once the holder releases mid-scan (stop), put the remaining
// messages back in original order, returning to plain FIFO.
const keep: OmniMessage[] = [];
let progressed = false;
for (let i = 0; i < this.pending.length; i++) {
if (this.promptActive || this.holder === null) {
keep.push(...this.pending.slice(i));
break;
}
const msg = this.pending[i]!;
if (this.streamOwner(msg) === this.holder) {
if (this.isStop(msg)) this.holder = null;
this.renderNow(msg);
progressed = true;
} else {
keep.push(msg);
}
}
this.pending = keep;
if (!progressed) break; // no message from the holder in the queue: wait for it to arrive
}
} finally {
this.draining = false;
}
}
/** Actually renders one message (queue scheduling is already done by drain). */
private renderNow(msg: OmniMessage): void {
if (msg.origin && msg.origin.length > 0) {
this.handleNested(msg);
return;
}
// The timestamp of this task's first (non-session_meta) message = the start point for
// the stats-line elapsed time. session_meta can predate this turn by a long time (a
// session may sit idle for a day before the first question), so it is excluded,
// matching Web / Trace accounting.
if (this.taskFirstTsMs === null && msg.type !== "session_meta") {
const ms = Date.parse(msg.timestamp);
if (Number.isFinite(ms)) this.taskFirstTsMs = ms;
}
if (isModelMessage(msg)) {
const payload = msg.payload;
switch (payload.type) {
case "partial_text":
this.handlePartialText(payload as PartialTextPayload);
return;
case "partial_thinking":
this.handlePartialThinking(payload as PartialThinkingPayload);
return;
case "partial_tool_call":
this.handlePartialToolCall(payload as PartialToolCallPayload);
return;
case "partial_tool_call_output":
this.handlePartialToolOutput(payload as PartialToolCallOutputPayload);
return;
// Complete (non-streaming) model_msg is never rendered (including image_url/inline_*); the content has already been shown by partial_*.
default:
return;
}
}
if (isEventMessage(msg)) {
const payload = msg.payload;
if (payload.type === "token_usage") {
// Accumulate this task's usage, printed together when the task ends (endTask),
// not shown after every tool call/round.
const p = payload as TokenUsagePayload;
this.sessionTotal = p.session.total;
if (this.compactionActive) {
// Usage of a compaction request: staged first (final attribution depends on
// whether a normal request_end still follows in this turn), and accumulated
// into compactionTokens so the compaction-completion line can show "usage this
// time"; does not update context accounting (see the compactionActive comment).
this.pendingCompactionTokens += p.request.total;
this.compactionTokens += p.request.total;
} else {
this.taskTokens += p.request.total;
this.contextNow = p.request.total; // current context = total of the most recent normal request
this.hasUsage = true;
}
} else if (payload.type === "approval_decision") {
// The approval result has usually already been rendered in place at the
// approval callback (noteApprovalDecision, guaranteeing three consecutive
// lines); deduplicated here by key; falls back to rendering one line (without a
// pairing tag) if it wasn't rendered yet.
const p = payload as ApprovalDecisionPayload;
if (this.renderedDecisions.delete(this.callLineKey(p.tool_call_id))) return;
this.finishLine();
this.out.write(`${dim(this.t.approvalDecision(p.decision))}\n`);
this.lastLineKey = null;
} else if (payload.type === "abort") {
// Run ended (user interrupt / retries exhausted): clear any pending retry state so the next run doesn't mistakenly print a retry line.
this.pendingRetry = null;
this.reconnectRun = 0;
this.finishLine();
this.out.write(`${formatAbort(payload as AbortPayload, this.t)}\n`);
this.lastLineKey = null;
} else if (payload.type === "request_begin") {
// The previous request ended in timeout/malformed -> this request is a retry
// carrying <turn_retried>: printed when the retry **actually starts** (when
// retries are exhausted, there's no retry after the last failure, only an abort
// explaining why).
if (this.pendingRetry) {
this.reconnectRun += 1;
this.finishLine();
this.out.write(`${dim(this.t.reconnectLabel(this.pendingRetry, this.reconnectRun))}\n`);
this.lastLineKey = null;
this.pendingRetry = null;
}
} else if (payload.type === "request_end") {
const p = payload as RequestEndPayload;
if (!this.compactionActive) {
// A non-compaction request_end = the end of the turn so far: records the
// timestamp (the end point for elapsed time), and settles any previously
// staged compaction usage — reaching here means that compaction was followed
// by a normal Request in this turn (mid-turn compaction), so its usage is
// attributed to this turn.
const ms = Date.parse(msg.timestamp);
if (Number.isFinite(ms)) this.taskLastReqEndMs = ms;
if (this.pendingCompactionTokens > 0) {
this.taskTokens += this.pendingCompactionTokens;
this.pendingCompactionTokens = 0;
this.hasUsage = true;
}
}
if (p.status === "timeout" || p.status === "malformed") {
this.pendingRetry = p.status;
} else {
this.pendingRetry = null;
this.reconnectRun = 0;
}
} else if (payload.type === "compaction_begin") {
// Paired compaction events: begin signals compaction is in progress.
const p = payload as CompactionBeginPayload;
this.finishLine();
this.compactionActive = true;
this.compactionTokens = 0;
this.out.write(`${dim(this.t.compactionStart(p.mode, p.reason))}\n`);
this.lastLineKey = null;
} else if (payload.type === "compaction_end") {
// end signals the result and shows the tokens consumed by the compaction request (if any).
const p = payload as CompactionEndPayload;
this.finishLine();
this.compactionActive = false;
// Same accounting as the stats line: total = Session cumulative (parent + child), delta = usage of this compaction.
const tokens =
this.compactionTokens > 0
? {
total: humanizeTokens(this.sessionTotal + this.subagentTotal),
delta: signedDelta(humanizeTokens(this.compactionTokens)),
}
: undefined;
this.compactionTokens = 0;
this.out.write(`${dim(this.t.compactionStop(p.mode, p.status, tokens))}\n`);
this.lastLineKey = null;
}
return;
}
// session_meta: not rendered.
}
/**
* Nested sub-session messages (carrying an origin): renders the child tool call
* (tagged `agent-xxx-tool-xxx` to mark it as coming from a subagent) and its approval
* result; the request delta of a child token_usage counts toward this task's usage;
* everything else is not rendered (see the rendering rule at the top of this file).
*/
private handleNested(msg: OmniMessage): void {
const origin = msg.origin!;
if (isModelMessage(msg)) {
if (msg.payload.type === "tool_call") {
// A complete tool_call renders one line (nested messages never render
// partial_*, so there's no duplication); one already rendered in place at
// approval time (message arrived later than the approval callback) is
// deduplicated by key and skipped.
const p = msg.payload as ToolCallPayload;
const key = this.callLineKey(p.tool_call_id, origin);
if (this.ensuredCallLines.has(key)) return;
this.renderCallLine(p, origin, key);
}
return;
}
if (isEventMessage(msg)) {
if (msg.payload.type === "approval_decision") {
// The approval result is usually already rendered in place at the approval callback; deduplicated here by key; falls back to rendering if it wasn't rendered yet.
const p = msg.payload as ApprovalDecisionPayload;
if (this.renderedDecisions.delete(this.callLineKey(p.tool_call_id, origin))) {
return;
}
this.finishLine();
this.out.write(`${dim(this.t.approvalDecision(p.decision))}\n`);
this.lastLineKey = null;
} else if (msg.payload.type === "token_usage") {
// Child-session usage counts toward this task's Token delta and the Session total (parent and child use the same accounting); context still follows parent-session accounting.
const req = (msg.payload as TokenUsagePayload).request.total;
this.taskTokens += req;
this.subagentTotal += req;
this.hasUsage = true;
}
}
}
private handlePartialText(p: PartialTextPayload): void {
if (p.event_type === "stop") {
this.finishLine();
return;
}
// Insert a line break when switching from thinking (dim) to body text, to avoid them running together.
if (this.inDim) this.finishLine();
if (p.text) {
this.out.write(p.text);
this.inLine = true;
this.lastLineKey = null;
}
}
private handlePartialThinking(p: PartialThinkingPayload): void {
if (p.event_type === "stop") {
this.finishLine();
return;
}
if (!this.inDim) {
this.out.write(DIM);
this.inDim = true;
}
if (p.thinking) {
this.out.write(p.thinking);
this.inLine = true;
this.lastLineKey = null;
}
}
private handlePartialToolCall(p: PartialToolCallPayload): void {
// The call line was already rendered in place from the complete message at approval time: skip the whole late-arriving streaming copy (clean up the buffer on stop).
if (this.ensuredCallLines.has(this.callLineKey(p.tool_call_id))) {
if (p.event_type === "stop") this.partialToolCalls.delete(p.tool_call_id);
return;
}
let partial = this.partialToolCalls.get(p.tool_call_id);
if (!partial) {
if (p.event_type === "stop") return;
partial = { name: p.name, arguments: "", lastPreview: "" };
this.partialToolCalls.set(p.tool_call_id, partial);
}
if (p.name) partial.name = p.name;
if (p.arguments) {
partial.arguments += p.arguments;
}
if (p.event_type === "stop") {
if (partial.lastPreview) this.finishLine();
this.partialToolCalls.delete(p.tool_call_id);
return;
}
if (!p.arguments) return;
if (this.inDim) this.finishLine();
const preview = renderPartialToolCall(partial.name, partial.arguments);
if (preview === null) return;
// The line starts with a pairing tag [tool-<last 3 chars of id>], matching the output line that follows.
const key = this.callLineKey(p.tool_call_id);
if (this.partialToolCallLineId !== p.tool_call_id) {
this.finishLine();
this.partialToolCallLineId = p.tool_call_id;
this.out.write(cyan(`[${callTag(p.tool_call_id)}] ${preview}`));
} else if (preview.startsWith(partial.lastPreview)) {
this.out.write(cyan(preview.slice(partial.lastPreview.length)));
} else {
// The preview usually grows monotonically with the arguments; if escaping/folding makes it non-appendable, start a new line with the current readable state.
this.finishLine();
this.partialToolCallLineId = p.tool_call_id;
this.out.write(cyan(`[${callTag(p.tool_call_id)}] ${preview}`));
}
partial.lastPreview = preview;
this.inLine = true;
this.lastLineKey = key;
}
private handlePartialToolOutput(p: PartialToolCallOutputPayload): void {
if (p.event_type === "stop") {
this.finishLine();
return;
}
if (this.inDim) this.finishLine();
if (p.output) this.writeToolOutput(p.output, callTag(p.tool_call_id));
// Image delta (carried whole in a single delta): the terminal doesn't render the
// image itself, so print one placeholder line per image, using the same pairing tag
// as the output gutter.
if (p.images && p.images.length > 0) {
this.finishLine();
const tag = callTag(p.tool_call_id);
for (const _ of p.images) {
this.out.write(`${DIM}[${tag}] >> [image]${RESET}\n`);
}
this.lastLineKey = null;
}
}
/**
* Writes tool-call **output** line by line, each line starting with the dim gutter
* `[tool-<last 3 chars of id>] >> `, paired with the call line (cyan `[tool-xxx] $
* cmd`). Streaming chunks arrive incrementally; whether to write the gutter is
* decided by the current line-start state.
*/
private writeToolOutput(chunk: string, tag: string): void {
let i = 0;
while (i < chunk.length) {
if (this.toolOutLineStart) {
this.out.write(`${DIM}[${tag}] >> ${RESET}`);
this.toolOutLineStart = false;
this.inLine = true;
}
const nl = chunk.indexOf("\n", i);
if (nl === -1) {
this.out.write(chunk.slice(i));
i = chunk.length;
} else {
this.out.write(chunk.slice(i, nl + 1));
this.toolOutLineStart = true;
this.inLine = false;
i = nl + 1;
}
}
this.lastLineKey = null;
}
/**
* Task end: forcibly releases the screen lock and drains any remaining messages
* (normally every streaming segment has already closed), finishes the current line,
* and prints one line of stats — all as Session cumulative values + this task's
* delta: context (input+output of the most recent request; delta = minus the context
* at the start of this task, which can be negative once compaction shrinks context),
* Token (Session cumulative = parent-session cumulative + child-session cumulative;
* delta = added this task, same accounting for parent and child), elapsed time
* (Session total elapsed; delta = this task's elapsed). This task's counters are then
* reset.
*/
endTask(elapsedMs = 0): void {
this.promptActive = false;
this.promptKey = null;
this.flushDeferredDecisions();
this.holder = null;
this.drain();
this.finishLine();
// This task's elapsed time = first message -> last non-compaction request_end
// (mid-turn compaction falls within the span and is counted; compaction after the
// turn ends falls after it and isn't). The degenerate case of a turn with no
// request_end at all (e.g. aborted before the first Request even ran) falls back to
// the externally supplied wall-clock elapsedMs. Any staged but unsettled compaction
// usage is discarded here (compaction after the turn ended isn't attributed to it).
const elapsed =
this.taskFirstTsMs !== null && this.taskLastReqEndMs !== null
? Math.max(0, this.taskLastReqEndMs - this.taskFirstTsMs)
: elapsedMs;
this.sessionElapsedMs += elapsed;
if (this.hasUsage) {
const contextDelta = this.contextNow - this.contextAtTaskStart;
this.out.write(
`${dim(
this.t.taskStats({
context: humanizeTokens(this.contextNow),
contextDelta: signedDelta(humanizeTokens(contextDelta)),
tokens: humanizeTokens(this.sessionTotal + this.subagentTotal),
tokensDelta: signedDelta(humanizeTokens(this.taskTokens)),
elapsed: humanizeDuration(this.sessionElapsedMs),
elapsedDelta: signedDelta(humanizeDuration(elapsed)),
}),
)}\n`,
);
this.contextAtTaskStart = this.contextNow;
this.lastLineKey = null;
}
this.taskTokens = 0;
this.pendingCompactionTokens = 0;
this.taskFirstTsMs = null;
this.taskLastReqEndMs = null;
this.hasUsage = false;
// Compaction always closes within run/compact (stop is always reached); this is a
// defensive reset to prevent state from leaking into the next task on an
// exceptional path.
this.compactionActive = false;
this.compactionTokens = 0;
// Dedup/buffer registrations are only meaningful within this task: clear them to prevent unbounded growth in long sessions (chat).
this.ensuredCallLines.clear();
this.renderedDecisions.clear();
this.partialToolCalls.clear();
}
/**
* Cleans up after a manual `/compact` (outside a Task boundary): compaction usage has
* already been shown on the compaction-completion line and counted into the Session
* total, so no stats line is printed here; only settles the Session elapsed time and
* resets this task's counters — otherwise the compaction's usage would remain in
* taskTokens and be mistakenly counted into the next task's `[stats]` delta (or never
* settled at all if the user exits right after).
*/
endCompact(elapsedMs = 0): void {
this.sessionElapsedMs += elapsedMs;
this.taskTokens = 0;
this.pendingCompactionTokens = 0;
this.taskFirstTsMs = null;
this.taskLastReqEndMs = null;
this.hasUsage = false;
this.compactionActive = false;
this.compactionTokens = 0;
}
private closeDim(): void {
if (this.inDim) {
this.out.write(RESET);
this.inDim = false;
}
}
/** Finishes the current streaming line: closes dim mode, emits a trailing newline, and resets tool output to line-start. */
private finishLine(): void {
this.closeDim();
if (this.inLine) {
this.out.write("\n");
this.inLine = false;
}
this.toolOutLineStart = true;
this.partialToolCallLineId = null;
}
}
+107
View File
@@ -0,0 +1,107 @@
/**
* Consumption loop that drives a Task to completion (CLI side, shared by run and chat).
*
* New protocol: `session.run(prompt, { signal, approve })` runs the entire ReAct loop in one
* call — within a turn, the engine invokes the `approve` callback for each tool_call, executing
* it on allow, with execution possibly overlapping. The CLI only needs to consume the output
* stream and supply `approve`. The approval strategy is determined by the permission mode
* (allow-all / deny-all / read-only / always-ask per-call approval).
*/
import { isEventMessage } from "@prismshadow/penguin-core";
import type { ApproveFn, OmniMessage, Session } from "@prismshadow/penguin-core";
import type { StreamRenderer } from "./render.js";
import { makeApprove, promptApproval, type ApprovalMode } from "./approval.js";
import type { Messages } from "./i18n.js";
export interface RunTaskOptions {
/** Approval mode (default allow-all). */
mode?: ApprovalMode;
/** Interrupt signal (Ctrl-C, etc.). */
signal?: AbortSignal;
renderer: StreamRenderer;
/** The actual Q&A for interactive approval; defaults to the one-off `promptApproval`. */
interactivePrompt?: ApproveFn;
/** Message set. */
t: Messages;
}
/** Result of one Task: `aborted` = the Task ended with an abort event (LLM failure/reconnect exhausted/user interrupt). */
export interface RunTaskResult {
aborted: boolean;
}
export async function runTask(
session: Session,
prompt: OmniMessage[],
opts: RunTaskOptions,
): Promise<RunTaskResult> {
const basePrompt: ApproveFn = opts.interactivePrompt ?? (() => promptApproval({ t: opts.t }));
// Lock the renderer while waiting for the user's approval input: messages from concurrent
// tools/subsessions are queued and released together once the Q&A finishes, so the prompt
// isn't scrambled by later output. The pending tool_call is passed in so its call line stays
// right before the prompt; the approval result is rendered in place **before unlocking** —
// "tool call → approval prompt → approval result" stays three consecutive lines, for both
// the main Agent and subagents (messages arriving via the async pipeline may lag behind the
// approval callback, hence render-in-place plus de-duplication of the copy).
//
// Serialization: the parent session and a run_subagent child session share this callback and
// may request approval concurrently (the parent is waiting on one approval while an
// already-approved child session starts its own). Concurrent prompts would clobber the same
// Q&A state and fight over the same stdin (one answer resolving two questions, leaving the
// other permanently stuck); a promise chain queues them so only one question is asked at a
// time.
let promptChain: Promise<unknown> = Promise.resolve();
const interactivePrompt: ApproveFn = (tc) => {
const result = promptChain.then(async () => {
opts.renderer.beginUserPrompt(tc);
try {
const decision = await basePrompt(tc);
opts.renderer.noteApprovalDecision(tc, decision);
return decision;
} finally {
opts.renderer.endUserPrompt();
}
});
promptChain = result.then(
() => undefined,
() => undefined,
);
return result;
};
const approveByMode = makeApprove({
mode: opts.mode ?? "allow-all",
toolPermission: (name) => session.toolPermission(name),
interactivePrompt,
});
// The auto-approval path (allow-all / deny-all / read-only approvals) has no prompt: it
// likewise renders the "call line → approval result" pair in place; the interactive path's
// already-rendered copy is idempotently de-duplicated inside note.
const approve: ApproveFn = async (tc) => {
const decision = await approveByMode(tc);
opts.renderer.noteApprovalDecision(tc, decision);
return decision;
};
// A single run drives the whole ReAct loop (the engine requests approval per call and runs
// tools concurrently within a turn). Once the task ends (including on error), endTask
// prints this task's stats (context/Token/elapsed time). The engine collapses failures
// (auth errors, reconnect exhausted, etc.) into a main-session abort event rather than
// throwing; the result reported here reflects that, for `penguin run` to map to
// an exit code.
const startedAt = Date.now();
let aborted = false;
try {
for await (const msg of session.run(prompt, {
approve,
...(opts.signal ? { signal: opts.signal } : {}),
})) {
if (isEventMessage(msg) && msg.payload.type === "abort" && (msg.origin?.length ?? 0) === 0) {
aborted = true;
}
opts.renderer.handle(msg);
}
} finally {
opts.renderer.endTask(Date.now() - startedAt);
}
return { aborted };
}
+151
View File
@@ -0,0 +1,151 @@
/**
* Streaming tool-call rendering (CLI side).
*
* The CLI only consumes `partial_tool_call` for visible rendering. exec_command is shown as
* `$ <cmd>` as early as possible; input_command / input_subagent show the target session id,
* with a non-empty payload (chars / prompt) appended as `<< <content>` — the payload is
* critical for approval and later audit (writing to stdin is equivalent to running a command),
* so the session id alone is not enough; run_subagent shows the prompt; other tools fall back
* to `name(args-prefix)`.
*
* The render layer streams by appending to the preview (see render.ts), so the preview format
* must stay append-only: rendering only starts once the target id has fully appeared, the
* payload is only appended at the end, and the preview stops growing once it hits the
* truncation limit.
*/
/** Max length of the single-line preview for a payload (chars / prompt); truncated with an ellipsis beyond this, after which the preview stops growing. */
const MAX_PAYLOAD_PREVIEW = 120;
/** Collapse to a single line: newlines/runs of whitespace become a single space, and leading/trailing whitespace is trimmed. */
function toSingleLine(text: string): string {
return text.replace(/\s+/g, " ").trim();
}
/**
* Turn control characters into a visible, faithful form so stdin writes don't garble the
* screen: `\n`/`\r`/`\t` are shown as escape literals (whether Enter was pressed is important
* information and must not collapse into a space), other C0 control chars and DEL use caret
* notation (U+0003 → `^C`); backslash itself is escaped to avoid ambiguity.
*/
function visualizeControlChars(text: string): string {
return text.replace(/[\\\u0000-\u001f\u007f]/g, (ch) => {
if (ch === "\\") return "\\\\";
if (ch === "\n") return "\\n";
if (ch === "\r") return "\\r";
if (ch === "\t") return "\\t";
if (ch === "\u007f") return "^?";
return `^${String.fromCharCode(ch.charCodeAt(0) + 64)}`;
});
}
/** Truncate to the single-line preview limit, appending an ellipsis if exceeded. */
function capPreview(text: string): string {
return text.length > MAX_PAYLOAD_PREVIEW ? `${text.slice(0, MAX_PAYLOAD_PREVIEW)}…` : text;
}
/** Extract the current value of a string field from a possibly-incomplete JSON object string. */
function extractPartialStringField(argsJson: string, field: string): string | null {
const key = `"${field}"`;
const keyIndex = argsJson.indexOf(key);
if (keyIndex === -1) return null;
let i = keyIndex + key.length;
while (/\s/.test(argsJson[i] ?? "")) i += 1;
if (argsJson[i] !== ":") return null;
i += 1;
while (/\s/.test(argsJson[i] ?? "")) i += 1;
if (argsJson[i] !== '"') return null;
i += 1;
let out = "";
let escaped = false;
for (; i < argsJson.length; i += 1) {
const ch = argsJson[i]!;
if (escaped) {
switch (ch) {
case "n":
out += "\n";
break;
case "r":
out += "\r";
break;
case "t":
out += "\t";
break;
case "b":
out += "\b";
break;
case "f":
out += "\f";
break;
case '"':
case "\\":
case "/":
out += ch;
break;
case "u": {
// If \uXXXX is cut off at an incremental chunk boundary, return "as far as we got":
// emitting the incomplete hex as a literal would cause a rollback once the next
// increment completes it (breaking append-only preview); the render layer falls
// back to a new line in that case.
if (i + 5 > argsJson.length) return out;
const hex = argsJson.slice(i + 1, i + 5);
if (/^[0-9a-fA-F]{4}$/.test(hex)) {
out += String.fromCharCode(Number.parseInt(hex, 16));
i += 4;
}
break;
}
default:
out += ch;
break;
}
escaped = false;
continue;
}
if (ch === "\\") {
escaped = true;
continue;
}
if (ch === '"') return out;
out += ch;
}
return out;
}
/**
* Streaming argument preview: exec_command shows `$ <cmd>` once cmd can be read; input_command /
* input_subagent show `⌨ <name> → <id>` once the target id is available, with a non-empty
* chars / prompt appended as `<< <content>` (an empty payload just means polling, left as-is);
* run_subagent shows `run_subagent << <prompt>` once prompt can be read; other tools fall back
* to name(args-prefix).
*/
export function renderPartialToolCall(name: string, argsJson: string): string | null {
if (!argsJson) return null;
if (name === "exec_command") {
const cmd = extractPartialStringField(argsJson, "cmd");
if (cmd !== null) return `$ ${toSingleLine(cmd)}`;
return null;
}
if (name === "run_subagent") {
const prompt = extractPartialStringField(argsJson, "prompt");
if (prompt !== null) return `run_subagent << ${capPreview(toSingleLine(prompt))}`;
return null;
}
if (name === "input_command") {
const pid = extractPartialStringField(argsJson, "process_id");
if (pid === null) return null;
const chars = extractPartialStringField(argsJson, "chars");
const payload = chars ? ` << ${capPreview(visualizeControlChars(chars))}` : "";
return `⌨ input_command → ${toSingleLine(pid)}${payload}`;
}
if (name === "input_subagent") {
const sid = extractPartialStringField(argsJson, "subagent_id");
if (sid === null) return null;
const prompt = extractPartialStringField(argsJson, "prompt");
const payload = prompt ? ` << ${capPreview(toSingleLine(prompt))}` : "";
return `⌨ input_subagent → ${toSingleLine(sid)}${payload}`;
}
return `${name || "tool_call"}(${toSingleLine(argsJson)}`;
}
+162
View File
@@ -0,0 +1,162 @@
import { describe, expect, it } from "vitest";
import { Readable, Writable } from "node:stream";
import { toolCall } from "@prismshadow/penguin-core";
import type { OmniMessage, ToolCallPayload } from "@prismshadow/penguin-core";
import { makeApprove, promptApproval, resolveApprovalMode } from "../src/approval.js";
import { getMessages } from "../src/i18n.js";
const t = getMessages("en");
/** An in-memory writable stream that collects everything written to output. */
function collector(): { stream: Writable; text: () => string } {
let buf = "";
const stream = new Writable({
write(chunk, _enc, cb) {
buf += chunk.toString();
cb();
},
});
return { stream, text: () => buf };
}
// mock_read_only_tool is only used for approval-mode tests, not a real tool; its permission is read-only ("r").
const readTool = (): OmniMessage<ToolCallPayload> =>
toolCall({ name: "mock_read_only_tool", arguments: '{"path":"a"}', toolCallId: "r1" });
const writeTool = (): OmniMessage<ToolCallPayload> =>
toolCall({ name: "exec_command", arguments: '{"cmd":"rm x"}', toolCallId: "w1" });
const perms: Record<string, "r" | "rw"> = {
mock_read_only_tool: "r",
exec_command: "rw",
};
const toolPermission = (name: string): "r" | "rw" | undefined => perms[name];
describe("promptApproval", () => {
it('returns "allow" when the user types "y"', async () => {
const { stream, text } = collector();
const decision = await promptApproval({
input: Readable.from(["y\n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
// Output is exactly the approval prompt itself — no input echo, no repeated tool-call rendering.
expect(text()).toBe("? Approve this tool call? [Y/n] ");
});
it('returns "allow" on empty input (Enter) — tool approval defaults to yes', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from(["\n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
});
it('returns "allow" for "yes" (case-insensitive, trimmed)', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from([" YES \n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
});
it('returns "deny" when the user types "n"', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from(["n\n"]),
output: stream,
t,
});
expect(decision).toBe("deny");
});
it('returns "allow" for unrelated input (tool approval defaults to yes)', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from(["maybe\n"]),
output: stream,
t,
});
expect(decision).toBe("allow");
});
it('returns "deny" when the input stream ends (EOF) instead of hanging', async () => {
const { stream } = collector();
const decision = await promptApproval({
input: Readable.from([]),
output: stream,
t,
});
expect(decision).toBe("deny");
});
});
describe("resolveApprovalMode", () => {
it("maps --approve values; defaults to allow-all", () => {
expect(resolveApprovalMode("allow-all", t)).toBe("allow-all");
expect(resolveApprovalMode("read-only", t)).toBe("read-only");
expect(resolveApprovalMode("deny-all", t)).toBe("deny-all");
expect(resolveApprovalMode("always-ask", t)).toBe("always-ask");
expect(resolveApprovalMode("READ-ONLY", t)).toBe("read-only");
expect(resolveApprovalMode(undefined, t)).toBe("allow-all");
});
});
describe("makeApprove permission modes", () => {
it("allow-all → allows everything", async () => {
const approve = makeApprove({
mode: "allow-all",
toolPermission,
interactivePrompt: async () => "deny",
});
expect(await approve(readTool())).toBe("allow");
expect(await approve(writeTool())).toBe("allow");
});
it("deny-all → rejects everything", async () => {
const approve = makeApprove({
mode: "deny-all",
toolPermission,
interactivePrompt: async () => "allow",
});
expect(await approve(readTool())).toBe("deny");
expect(await approve(writeTool())).toBe("deny");
});
it("read-only → auto-allows read-only tools, prompts for the rest", async () => {
let prompted = 0;
const approve = makeApprove({
mode: "read-only",
toolPermission,
interactivePrompt: async () => {
prompted += 1;
return "deny";
},
});
// Read-only tools are auto-allowed without prompting.
expect(await approve(readTool())).toBe("allow");
expect(prompted).toBe(0);
// Read-write tools are handed off to the interactive prompt (denied here).
expect(await approve(writeTool())).toBe("deny");
expect(prompted).toBe(1);
});
it("always-ask → always delegates to the interactive prompt", async () => {
let prompted = 0;
const approve = makeApprove({
mode: "always-ask",
toolPermission,
interactivePrompt: async () => {
prompted += 1;
return "allow";
},
});
expect(await approve(readTool())).toBe("allow");
expect(await approve(writeTool())).toBe("allow");
expect(prompted).toBe(2);
});
});
+45
View File
@@ -0,0 +1,45 @@
import { describe, expect, it } from "vitest";
import { decideSigint } from "../src/commands/chat.js";
import { parseApprovalAnswer } from "../src/approval.js";
describe("decideSigint (Ctrl-C 行为状态机)", () => {
it("approving → deny(无论缓冲区是否有内容)", () => {
expect(decideSigint("approving", false)).toBe("deny");
expect(decideSigint("approving", true)).toBe("deny");
});
it("running → abort(中断当前 Task,不退出)", () => {
expect(decideSigint("running", false)).toBe("abort");
expect(decideSigint("running", true)).toBe("abort");
});
it("idle + 有输入 → clear(清空缓冲区)", () => {
expect(decideSigint("idle", true)).toBe("clear");
});
it("idle + 无输入 → confirm-exit(弹出 y/N 退出确认)", () => {
expect(decideSigint("idle", false)).toBe("confirm-exit");
});
it("confirming-exit → exit(确认中再次 Ctrl-C 直接退出)", () => {
expect(decideSigint("confirming-exit", false)).toBe("exit");
expect(decideSigint("confirming-exit", true)).toBe("exit");
});
});
describe("parseApprovalAnswer", () => {
it("y / yes(trim、不区分大小写)→ allow;n / no → deny", () => {
expect(parseApprovalAnswer("y")).toBe("allow");
expect(parseApprovalAnswer(" YES \n")).toBe("allow");
expect(parseApprovalAnswer("Y")).toBe("allow");
expect(parseApprovalAnswer("n")).toBe("deny");
expect(parseApprovalAnswer("NO")).toBe("deny");
});
it("空/无关输入用 fallback(缺省 deny;工具审批传 allow)", () => {
expect(parseApprovalAnswer("")).toBe("deny"); // default fallback
expect(parseApprovalAnswer("nope")).toBe("deny");
expect(parseApprovalAnswer("", "allow")).toBe("allow"); // tool approval defaults to allow
expect(parseApprovalAnswer("nope", "allow")).toBe("allow");
expect(parseApprovalAnswer("n", "allow")).toBe("deny"); // explicit n still denies
});
});
+68
View File
@@ -0,0 +1,68 @@
/**
* Unit tests for `config model list` rendering: provider and model_id are separate
* columns (stored fields as-is, with the default model marked `*` before the provider
* column; the request column was removed along with concatenated storage); vision falls
* back to the catalog matched by the (provider, model_id) pair; api_key is masked
* inline; fully empty columns are omitted automatically.
*/
import { describe, expect, it } from "vitest";
import type { ProjectConfig } from "@prismshadow/penguin-core";
import { formatModelRows } from "../src/commands/config.js";
describe("formatModelRows", () => {
const cfg: ProjectConfig = {
default_model: { provider: "anthropic", model_id: "claude-sonnet-4-6" },
models: [
{
provider: "anthropic",
model_id: "claude-sonnet-4-6",
context_window: 1000000,
pricing: { unit: "usd_per_mtok", cache_read: 0.3, cache_write: 3.75, output: 15 },
},
{
provider: "custom",
model_id: "my-proxy-model",
client_type: "openai",
vision: false,
api_key: "sk-test-abcd-1234",
},
],
};
it("provider 与 model_id 双列展示;预置模型 vision 经目录成对匹配,默认模型以 * 标记", () => {
const lines = formatModelRows(cfg);
expect(lines).toHaveLength(2);
expect(lines[0]).toMatch(/^\* anthropic\s+claude-sonnet-4-6\s+vision=Y/);
expect(lines[0]).toContain("price=0.3/3.75/15");
// The request column was removed; no <provider>/<id> concatenation appears anymore.
expect(lines[0]).not.toContain("request=");
expect(lines[0]).not.toContain("anthropic/claude-sonnet-4-6");
});
it("自定义模型 vision 按标注(显式 false 记 -);内联 api_key 掩码显示", () => {
const lines = formatModelRows(cfg);
expect(lines[1]).toMatch(/^ {2}custom\s+my-proxy-model\s+vision=-/);
expect(lines[1]).toContain("client_type=openai");
expect(lines[1]).toContain("api_key=****1234");
expect(lines[1]).not.toContain("sk-test-abcd-1234");
});
it("同名 model_id 双 provider 并存时各占一行,默认标记只落在成对命中的那行", () => {
const lines = formatModelRows({
default_model: { provider: "deepseek", model_id: "m1" },
models: [
{ provider: "deepseek", model_id: "m1" },
{ provider: "siliconflow", model_id: "m1" },
],
});
expect(lines[0]).toMatch(/^\* deepseek\s+m1\s+vision=Y/);
expect(lines[1]).toMatch(/^ {2}siliconflow\s+m1\s+vision=Y/);
});
it("无标注按「缺省=支持」记 Y;全空列省略", () => {
const lines = formatModelRows({
models: [{ provider: "custom", model_id: "m1" }],
});
expect(lines[0]).toBe(" custom m1 vision=Y api_key=-");
});
});
+301
View File
@@ -0,0 +1,301 @@
/**
* Integration tests for `penguin config model add|default|vision|list` (run through
* commander's parseAsync for the full command path): --model-id always takes the
* upstream id, paired with --provider to form a (provider, model_id) reference (add's
* --provider defaults to catalog-based inference, falling back to custom when
* inference fails; default / vision require --provider and raise an error when the
* reference isn't found in models — no string concatenation is ever performed); --root
* specifies the data root directory (takes priority over PENGUIN_HOME); persisted to a
* single hidden .project_config.toml (mode 0600, credentials inline, provider and
* model_id as separate columns); list displays provider and model_id as separate
* columns.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { Command } from "commander";
import { parse as parseToml } from "smol-toml";
import { DEFAULT_PROJECT_ID, projectConfigPath } from "@prismshadow/penguin-core";
import { registerConfigCommand } from "../src/commands/config.js";
import { getMessages } from "../src/i18n.js";
let tmpHome: string;
let tmpRoot: string;
let prevHome: string | undefined;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpHome = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-home-"));
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-root-"));
process.env.PENGUIN_HOME = tmpHome;
});
afterEach(async () => {
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
else process.env.PENGUIN_HOME = prevHome;
await fs.rm(tmpHome, { recursive: true, force: true });
await fs.rm(tmpRoot, { recursive: true, force: true });
});
interface TomlModelRef {
provider: string;
model_id: string;
}
/**
* Runs a `penguin config model …` command, capturing stdout / stderr and the exit code
* (without actually exiting the process; under exitOverride, commander usage errors —
* such as a missing required option — are thrown as a CommanderError, which is
* converted to a non-zero exit code).
*/
async function runModel(args: string[]): Promise<{ out: string; err: string; code: number }> {
const program = new Command();
program.exitOverride();
registerConfigCommand(program, getMessages("en"));
const out: string[] = [];
const err: string[] = [];
const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => {
out.push(String(chunk));
return true;
});
const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
err.push(String(chunk));
return true;
});
const prevExitCode = process.exitCode;
process.exitCode = undefined;
try {
await program.parseAsync(["node", "penguin", "config", "model", ...args]);
return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) };
} catch (e) {
const exitCode = (e as { exitCode?: number }).exitCode;
return { out: out.join(""), err: err.join(""), code: exitCode || 1 };
} finally {
outSpy.mockRestore();
errSpy.mockRestore();
process.exitCode = prevExitCode;
}
}
describe("penguin config model add/list(--root 与 provider / model_id 分列存储)", () => {
it("--root 优先于 PENGUIN_HOME:落盘到指定根目录的隐藏 .project_config.toml(0600)", async () => {
const add = await runModel([
"add",
"--model-id",
"my-own-model",
"--api-key",
"sk-root-secret-1",
"--root",
tmpRoot,
]);
expect(add.code).toBe(0);
// Catalog inference fails -> falls back to the custom group (provider is a separate field, never concatenated into the id).
expect(add.out).toContain("Added model (provider=custom, model_id=my-own-model).");
const file = projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID);
expect(path.basename(file)).toBe(".project_config.toml");
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
const parsed = parseToml(await fs.readFile(file, "utf8")) as {
models: Array<Record<string, unknown>>;
};
const entry = parsed.models.find(
(m) => m.provider === "custom" && m.model_id === "my-own-model",
);
expect(entry).toBeDefined();
expect(entry?.api_key).toBe("sk-root-secret-1");
// Concatenated storage id and request_model_id have been removed.
expect(entry?.request_model_id).toBeUndefined();
// The root directory pointed to by PENGUIN_HOME is unaffected.
await expect(fs.access(projectConfigPath(tmpHome, DEFAULT_PROJECT_ID))).rejects.toThrow();
// list also reads --root: provider and model_id as separate columns + masked api_key (the request column has been removed).
const list = await runModel(["list", "--root", tmpRoot]);
expect(list.code).toBe(0);
const line = list.out.split("\n").find((l) => l.includes("my-own-model"));
expect(line).toMatch(/custom\s+my-own-model/);
expect(line).toContain("api_key=****et-1");
expect(list.out).not.toContain("request=");
expect(list.out).not.toContain("sk-root-secret-1");
});
it("内置目录推断分组:上游 id 命中目录时条目落该 provider;--set-default 写成对引用", async () => {
const add = await runModel([
"add",
"--model-id",
"claude-sonnet-4-6",
"--set-default",
"--root",
tmpRoot,
]);
expect(add.code).toBe(0);
expect(add.out).toContain("Updated model (provider=anthropic, model_id=claude-sonnet-4-6).");
expect(add.out).toContain("Default model: (provider=anthropic, model_id=claude-sonnet-4-6)");
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as unknown as { default_model: TomlModelRef; models: Array<Record<string, unknown>> };
expect(parsed.default_model).toEqual({
provider: "anthropic",
model_id: "claude-sonnet-4-6",
});
expect(
parsed.models.find((m) => m.provider === "anthropic" && m.model_id === "claude-sonnet-4-6"),
).toBeDefined();
});
it("--provider 显式指定分组:同名上游 id 与预置条目互不冲突(各自独立条目)", async () => {
const add = await runModel([
"add",
"--model-id",
"claude-sonnet-4-6",
"--provider",
"myproxy",
"--base-url",
"https://proxy.example/v1",
"--root",
tmpRoot,
]);
expect(add.code).toBe(0);
expect(add.out).toContain("Added model (provider=myproxy, model_id=claude-sonnet-4-6).");
const list = await runModel(["list", "--root", tmpRoot]);
const line = list.out.split("\n").find((l) => l.includes("myproxy"));
expect(line).toMatch(/myproxy\s+claude-sonnet-4-6/);
expect(line).toContain("base_url=https://proxy.example/v1");
// The pre-existing anthropic entry remains (the (provider, model_id) pair naturally disambiguates).
expect(list.out.split("\n").some((l) => /anthropic\s+claude-sonnet-4-6/.test(l))).toBe(true);
});
it("client_type 缺省按分组语义(PRN-021):custom / 自建 / 网关落 openai,一方厂商不落", async () => {
// custom (catalog inference fails) and self-hosted groups (--provider not a catalog value): default to client_type=openai.
await runModel(["add", "--model-id", "my-openai-proxy", "--root", tmpRoot]);
await runModel(["add", "--model-id", "in-house-1", "--provider", "mylab", "--root", tmpRoot]);
// A non-catalog id under a first-party vendor group: client_type is not set (AgentHub auto-routes by upstream id).
await runModel([
"add",
"--model-id",
"my-fine-tune",
"--provider",
"deepseek",
"--root",
tmpRoot,
]);
// Gateway group: openai + the gateway's endpoint base URL pre-filled.
await runModel([
"add",
"--model-id",
"acme/some-model",
"--provider",
"openrouter",
"--root",
tmpRoot,
]);
// An explicit --client-type is persisted as-is, not overridden by the default rule.
await runModel([
"add",
"--model-id",
"special-1",
"--provider",
"mylab",
"--client-type",
"verbatim-type",
"--root",
tmpRoot,
]);
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as { models: Array<Record<string, unknown>> };
const by = (p: string, id: string) =>
parsed.models.find((m) => m.provider === p && m.model_id === id)!;
expect(by("custom", "my-openai-proxy").client_type).toBe("openai");
expect(by("mylab", "in-house-1").client_type).toBe("openai");
expect(by("deepseek", "my-fine-tune").client_type).toBeUndefined();
expect(by("openrouter", "acme/some-model").client_type).toBe("openai");
expect(by("openrouter", "acme/some-model").base_url).toBe("https://openrouter.ai/api/v1");
expect(by("mylab", "special-1").client_type).toBe("verbatim-type");
});
it("model default 经 --root 指定根目录设置默认模型(--model-id 上游 id + --provider 成对)", async () => {
const set = await runModel([
"default",
"--model-id",
"deepseek-v4-flash",
"--provider",
"deepseek",
"--root",
tmpRoot,
]);
expect(set.code).toBe(0);
expect(set.out).toContain(
"Default model set to (provider=deepseek, model_id=deepseek-v4-flash).",
);
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as unknown as { default_model: TomlModelRef };
expect(parsed.default_model).toEqual({
provider: "deepseek",
model_id: "deepseek-v4-flash",
});
});
});
describe("model default/vision:--provider 必填,(provider, model_id) 成对引用", () => {
it("缺 --provider:commander 用法报错,非零退出码", async () => {
const bad = await runModel(["default", "--model-id", "deepseek-v4-flash", "--root", tmpRoot]);
expect(bad.code).not.toBe(0);
expect(bad.err).toContain("--provider");
});
it("引用落空:成对引用不在 models 中,报错带成对引用与 model list 提示", async () => {
const bad = await runModel([
"default",
"--model-id",
"no-such-model",
"--provider",
"custom",
"--root",
tmpRoot,
]);
expect(bad.code).toBe(1);
expect(bad.err).toContain("(provider=custom, model_id=no-such-model)");
expect(bad.err).toContain("penguin config model list");
// The upstream id matches a pre-existing entry but --provider names the wrong group: also not found (exact pair, no fuzzy matching).
const wrongGroup = await runModel([
"vision",
"--model-id",
"claude-sonnet-4-6",
"--provider",
"openai",
"--root",
tmpRoot,
]);
expect(wrongGroup.code).toBe(1);
expect(wrongGroup.err).toContain("(provider=openai, model_id=claude-sonnet-4-6)");
expect(wrongGroup.err).toContain("penguin config model list");
});
it("model vision 成对引用命中:设置视觉模型(落盘内联表)", async () => {
const ok = await runModel([
"vision",
"--model-id",
"claude-sonnet-4-6",
"--provider",
"anthropic",
"--root",
tmpRoot,
]);
expect(ok.code).toBe(0);
expect(ok.out).toContain(
"Vision model set to (provider=anthropic, model_id=claude-sonnet-4-6).",
);
const parsed = parseToml(
await fs.readFile(projectConfigPath(tmpRoot, DEFAULT_PROJECT_ID), "utf8"),
) as unknown as { vision_model: TomlModelRef };
expect(parsed.vision_model).toEqual({
provider: "anthropic",
model_id: "claude-sonnet-4-6",
});
});
});
+141
View File
@@ -0,0 +1,141 @@
/**
* Integration tests for `penguin config vault set|list|remove` (run through commander's
* parseAsync for the full command path, with PENGUIN_HOME pointed at a temp directory):
* writes to a hidden .vault.toml (mode 0600), list masks values without leaking
* plaintext, remove raises an error on a missing key, --agent-id targets a specific
* Agent, and an invalid key name / an overlong value exit with a non-zero code.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { Command } from "commander";
import { agentVaultPath, DEFAULT_PROJECT_ID } from "@prismshadow/penguin-core";
import { registerConfigCommand } from "../src/commands/config.js";
import { getMessages } from "../src/i18n.js";
let tmpRoot: string;
let prevHome: string | undefined;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-vault-"));
process.env.PENGUIN_HOME = tmpRoot;
});
afterEach(async () => {
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
else process.env.PENGUIN_HOME = prevHome;
await fs.rm(tmpRoot, { recursive: true, force: true });
});
/** Runs a `penguin config vault …` command, capturing stdout/stderr and the exit code (without actually exiting the process). */
async function runVault(args: string[]): Promise<{ out: string; err: string; code: number }> {
const program = new Command();
program.exitOverride();
registerConfigCommand(program, getMessages("en"));
const out: string[] = [];
const err: string[] = [];
const outSpy = vi.spyOn(process.stdout, "write").mockImplementation((chunk) => {
out.push(String(chunk));
return true;
});
const errSpy = vi.spyOn(process.stderr, "write").mockImplementation((chunk) => {
err.push(String(chunk));
return true;
});
const prevExitCode = process.exitCode;
process.exitCode = undefined;
try {
await program.parseAsync(["node", "penguin", "config", "vault", ...args]);
return { out: out.join(""), err: err.join(""), code: Number(process.exitCode ?? 0) };
} finally {
outSpy.mockRestore();
errSpy.mockRestore();
process.exitCode = prevExitCode;
}
}
describe("penguin config vault", () => {
it("set → list(掩码)→ remove 全链路;落盘为隐藏 .vault.toml 且 0600", async () => {
const set = await runVault(["set", "--key", "MY_KEY", "--value", "vault-secret-9876"]);
expect(set.code).toBe(0);
expect(set.out).toContain("Saved vault entry MY_KEY.");
const file = agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, "default_agent");
expect(path.basename(file)).toBe(".vault.toml");
expect((await fs.stat(file)).mode & 0o777).toBe(0o600);
expect(await fs.readFile(file, "utf8")).toContain("vault-secret-9876");
const list = await runVault(["list"]);
expect(list.code).toBe(0);
expect(list.out).toContain("MY_KEY");
expect(list.out).toContain("****9876");
// Plaintext never appears in list output.
expect(list.out).not.toContain("vault-secret-9876");
const removed = await runVault(["remove", "--key", "MY_KEY"]);
expect(removed.code).toBe(0);
expect(removed.out).toContain("Removed vault entry MY_KEY.");
const empty = await runVault(["list"]);
expect(empty.out).toContain("The vault is empty.");
});
it("--agent-id 定向到目标 Agent 的 vault,不影响 default_agent", async () => {
const set = await runVault([
"set",
"--key",
"ONLY_A",
"--value",
"va-secret-value-1",
"--agent-id",
"agent-a",
]);
expect(set.code).toBe(0);
expect(
await fs.readFile(agentVaultPath(tmpRoot, DEFAULT_PROJECT_ID, "agent-a"), "utf8"),
).toContain("ONLY_A");
const defaultList = await runVault(["list"]);
expect(defaultList.out).toContain("The vault is empty.");
});
it("--root 指定数据根目录(优先于 PENGUIN_HOME),set/list 均定向到该根目录", async () => {
const otherRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-cli-vault-root-"));
try {
const set = await runVault([
"set",
"--key",
"ROOTED_KEY",
"--value",
"root-secret-value-1",
"--root",
otherRoot,
]);
expect(set.code).toBe(0);
expect(
await fs.readFile(agentVaultPath(otherRoot, DEFAULT_PROJECT_ID, "default_agent"), "utf8"),
).toContain("ROOTED_KEY");
// The root directory pointed to by PENGUIN_HOME is unaffected.
const defaultList = await runVault(["list"]);
expect(defaultList.out).toContain("The vault is empty.");
const rootedList = await runVault(["list", "--root", otherRoot]);
expect(rootedList.out).toContain("ROOTED_KEY");
} finally {
await fs.rm(otherRoot, { recursive: true, force: true });
}
});
it("非法键名 / 超长值以非零码退出并打印原因;remove 不存在的键报错", async () => {
const badKey = await runVault(["set", "--key", "1BAD", "--value", "v"]);
expect(badKey.code).toBe(1);
expect(badKey.err).toContain("Invalid vault key");
const tooLong = await runVault(["set", "--key", "OK_BIG", "--value", "x".repeat(8193)]);
expect(tooLong.code).toBe(1);
expect(tooLong.err).toContain("too long");
const ghost = await runVault(["remove", "--key", "GHOST"]);
expect(ghost.code).toBe(1);
expect(ghost.err).toContain("Vault entry GHOST does not exist.");
});
});
+69
View File
@@ -0,0 +1,69 @@
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import { getMessages, maskApiKey, resolveLanguage } from "../src/i18n.js";
describe("resolveLanguage (env PENGUIN_LANG, default en)", () => {
let prev: string | undefined;
beforeEach(() => {
prev = process.env.PENGUIN_LANG;
});
afterEach(() => {
if (prev === undefined) delete process.env.PENGUIN_LANG;
else process.env.PENGUIN_LANG = prev;
});
it("defaults to en when unset", () => {
delete process.env.PENGUIN_LANG;
expect(resolveLanguage()).toBe("en");
});
it("matches zh exactly (case-insensitive, trimmed)", () => {
process.env.PENGUIN_LANG = "zh";
expect(resolveLanguage()).toBe("zh");
process.env.PENGUIN_LANG = " ZH ";
expect(resolveLanguage()).toBe("zh");
});
it("falls back to en for non-exact zh prefixes and anything else", () => {
process.env.PENGUIN_LANG = "zh-CN"; // no longer prefix-matched -> en
expect(resolveLanguage()).toBe("en");
process.env.PENGUIN_LANG = "fr";
expect(resolveLanguage()).toBe("en");
process.env.PENGUIN_LANG = "en";
expect(resolveLanguage()).toBe("en");
});
});
describe("getMessages", () => {
it("provides zh and en runtime + help strings", () => {
expect(getMessages("zh").modelAdded("m", "m")).toContain("已添加");
expect(getMessages("en").modelAdded("m", "m")).toContain("Added");
expect(getMessages("zh").modelUpdated("m", "m")).toContain("已更新");
expect(getMessages("en").modelUpdated("m", "m")).toContain("Updated");
// Command/option descriptions are also localized.
expect(getMessages("zh").config.addDesc).toContain("模型");
expect(getMessages("en").config.addDesc).toContain("model");
expect(getMessages("en").run.desc).toContain("Task");
// config lang copy.
expect(getMessages("zh").config.langDesc).toContain("语言");
expect(getMessages("en").config.langDesc).toContain("language");
expect(getMessages("en").langSet("zh", "/x/.zshrc")).toContain("/x/.zshrc");
expect(getMessages("zh").langInvalid("fr")).toContain("fr");
});
it("header order is agent → workspace → model", () => {
const h = getMessages("en").header("run", "ag", "/ws", "mod");
expect(h.indexOf("agent=ag")).toBeLessThan(h.indexOf("workspace=/ws"));
expect(h.indexOf("workspace=/ws")).toBeLessThan(h.indexOf("model=mod"));
});
});
describe("maskApiKey", () => {
it("masks all but the last 4 chars", () => {
expect(maskApiKey("sk-1234567890")).toBe("****7890");
});
it("fully masks short keys (≤12 chars would leak most of the secret)", () => {
expect(maskApiKey("sk-test-1234")).toBe("***");
expect(maskApiKey("short")).toBe("***");
});
it("returns - when absent", () => {
expect(maskApiKey(undefined)).toBe("-");
});
});
+110
View File
@@ -0,0 +1,110 @@
import { describe, expect, it } from "vitest";
import {
LineComposer,
PasteFilter,
endsWithContinuation,
splitTrailingPartial,
} from "../src/input.js";
/** Feeds a series of input chunks into PasteFilter, collecting the forwarded output and paste events. */
async function runFilter(chunks: string[]): Promise<{ forwarded: string; pastes: string[] }> {
const filter = new PasteFilter();
const pastes: string[] = [];
let forwarded = "";
filter.on("data", (d: Buffer) => {
forwarded += d.toString("utf8");
});
filter.on("paste", (t: string) => pastes.push(t));
for (const c of chunks) filter.write(c);
await new Promise<void>((resolve) => {
filter.end(() => resolve());
});
return { forwarded, pastes };
}
describe("splitTrailingPartial", () => {
it("holds a trailing partial-marker prefix", () => {
expect(splitTrailingPartial("abc\x1b[200", "\x1b[200~")).toEqual({
emit: "abc",
hold: "\x1b[200",
});
});
it("holds nothing when no trailing prefix", () => {
expect(splitTrailingPartial("hello", "\x1b[200~")).toEqual({
emit: "hello",
hold: "",
});
});
});
describe("PasteFilter", () => {
it("forwards normal bytes unchanged", async () => {
const { forwarded, pastes } = await runFilter(["hello\r"]);
expect(forwarded).toBe("hello\r");
expect(pastes).toEqual([]);
});
it("strips markers and emits the pasted block (incl. newlines) as one event", async () => {
const { forwarded, pastes } = await runFilter(["\x1b[200~line1\nline2\nline3\x1b[201~"]);
expect(pastes).toEqual(["line1\nline2\nline3"]);
expect(forwarded).toBe(""); // pasted content is not forwarded to readline
});
it("keeps surrounding typed bytes and paste together in order", async () => {
const { forwarded, pastes } = await runFilter(["ab\x1b[200~PASTED\x1b[201~cd\r"]);
expect(forwarded).toBe("abcd\r");
expect(pastes).toEqual(["PASTED"]);
});
it("handles a marker split across chunks", async () => {
const { forwarded, pastes } = await runFilter(["x\x1b[20", "0~mid\x1b[201", "~y\r"]);
expect(forwarded).toBe("xy\r");
expect(pastes).toEqual(["mid"]);
});
});
describe("endsWithContinuation", () => {
it("odd trailing backslashes → continuation", () => {
expect(endsWithContinuation("foo\\")).toBe(true);
expect(endsWithContinuation("foo\\\\\\")).toBe(true);
});
it("even/none → not continuation", () => {
expect(endsWithContinuation("foo")).toBe(false);
expect(endsWithContinuation("foo\\\\")).toBe(false);
});
});
describe("LineComposer", () => {
it("single line → immediate message", () => {
const c = new LineComposer();
expect(c.pushTypedLine("hello")).toEqual({ message: "hello" });
});
it("backslash continuation joins lines with \\n", () => {
const c = new LineComposer();
expect(c.pushTypedLine("a\\")).toEqual({});
expect(c.pushTypedLine("b\\")).toEqual({});
expect(c.pushTypedLine("c")).toEqual({ message: "a\nb\nc" });
});
it("paste buffers a block, Enter on empty line sends it", () => {
const c = new LineComposer();
expect(c.pushPaste("l1\nl2\n")).toEqual({ lineCount: 2, normalized: "l1\nl2" });
expect(c.hasPending()).toBe(true);
expect(c.pushTypedLine("")).toEqual({ message: "l1\nl2" });
expect(c.hasPending()).toBe(false);
});
it("paste then typed text appends the text before sending", () => {
const c = new LineComposer();
c.pushPaste("l1\nl2");
expect(c.pushTypedLine("more")).toEqual({ message: "l1\nl2\nmore" });
});
it("reset clears pending", () => {
const c = new LineComposer();
c.pushPaste("a\nb");
c.reset();
expect(c.hasPending()).toBe(false);
});
});
+90
View File
@@ -0,0 +1,90 @@
import { mkdtemp, readFile, rm } from "node:fs/promises";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { afterEach, describe, expect, it } from "vitest";
import { applyLanguageToRc, resolveShellRc, upsertBlock } from "../src/lang-config.js";
describe("resolveShellRc", () => {
it("maps zsh / bash / fish to their startup files and syntax", () => {
const zsh = resolveShellRc("/bin/zsh", "/home/u");
expect(zsh.kind).toBe("zsh");
expect(zsh.rcPath).toBe("/home/u/.zshrc");
expect(zsh.body("zh")).toBe("export PENGUIN_LANG=zh");
const bash = resolveShellRc("/usr/bin/bash", "/home/u");
expect(bash.kind).toBe("bash");
expect(bash.rcPath).toBe("/home/u/.bashrc");
const fish = resolveShellRc("/usr/local/bin/fish", "/home/u");
expect(fish.kind).toBe("fish");
expect(fish.rcPath).toBe("/home/u/.config/fish/config.fish");
expect(fish.body("en")).toBe("set -gx PENGUIN_LANG en");
});
it("falls back to ~/.profile for an unknown shell", () => {
const rc = resolveShellRc(undefined, "/home/u");
expect(rc.kind).toBe("unknown");
expect(rc.rcPath).toBe("/home/u/.profile");
});
});
describe("upsertBlock", () => {
it("appends a marked block when none exists", () => {
const out = upsertBlock("export PATH=/x\n", "export PENGUIN_LANG=zh");
expect(out).toContain("export PATH=/x");
expect(out).toContain("# >>> PenguinHarness PENGUIN_LANG >>>");
expect(out).toContain("export PENGUIN_LANG=zh");
expect(out).toContain("# <<< PenguinHarness PENGUIN_LANG <<<");
});
it("replaces the block in place and is idempotent", () => {
const first = upsertBlock("", "export PENGUIN_LANG=zh");
const second = upsertBlock(first, "export PENGUIN_LANG=en");
// Only one block remains, with its content replaced by the latest value.
expect(second.match(/PenguinHarness PENGUIN_LANG/g)?.length).toBe(2); // begin + end markers
expect(second).toContain("export PENGUIN_LANG=en");
expect(second).not.toContain("export PENGUIN_LANG=zh");
// Writing the same value again is stable (the block does not keep growing).
const third = upsertBlock(second, "export PENGUIN_LANG=en");
expect(third).toBe(second);
});
it("preserves surrounding content when replacing", () => {
const base = "line1\n" + upsertBlock("", "export PENGUIN_LANG=zh") + "line2\n";
const out = upsertBlock(base, "export PENGUIN_LANG=en");
expect(out.startsWith("line1\n")).toBe(true);
expect(out.endsWith("line2\n")).toBe(true);
expect(out).toContain("export PENGUIN_LANG=en");
});
});
describe("applyLanguageToRc", () => {
let home: string;
afterEach(async () => {
await rm(home, { recursive: true, force: true });
});
it("writes the export line to the resolved startup file", async () => {
home = await mkdtemp(join(tmpdir(), "penguin-lang-"));
const { rcPath, kind } = await applyLanguageToRc("zh", { shell: "/bin/zsh", home });
expect(kind).toBe("zsh");
expect(rcPath).toBe(join(home, ".zshrc"));
const content = await readFile(rcPath, "utf8");
expect(content).toContain("export PENGUIN_LANG=zh");
// Switching the language again updates the file in place instead of appending.
await applyLanguageToRc("en", { shell: "/bin/zsh", home });
const updated = await readFile(rcPath, "utf8");
expect(updated).toContain("export PENGUIN_LANG=en");
expect(updated).not.toContain("export PENGUIN_LANG=zh");
expect(updated.match(/# >>> PenguinHarness/g)?.length).toBe(1);
});
it("creates nested config dir for fish", async () => {
home = await mkdtemp(join(tmpdir(), "penguin-lang-"));
const { rcPath } = await applyLanguageToRc("en", { shell: "/usr/bin/fish", home });
expect(rcPath).toBe(join(home, ".config", "fish", "config.fish"));
const content = await readFile(rcPath, "utf8");
expect(content).toContain("set -gx PENGUIN_LANG en");
});
});
+716
View File
@@ -0,0 +1,716 @@
import { describe, expect, it } from "vitest";
import { Writable } from "node:stream";
import {
approvalDecision,
abortEvent,
assistantText,
compactionBegin,
compactionEnd,
requestBegin,
requestEnd,
thinkingMessage,
toolCall,
toolCallOutput,
tokenUsage,
sessionMeta,
partialText,
partialThinking,
partialToolCall,
partialToolCallOutput,
withOrigin,
} from "@prismshadow/penguin-core";
import type { MessageOrigin } from "@prismshadow/penguin-core";
import { StreamRenderer, formatAbort, humanizeTokens, renderHistory } from "../src/render.js";
import { getMessages } from "../src/i18n.js";
const t = getMessages("en");
function collector(): { stream: Writable; text: () => string } {
let buf = "";
const stream = new Writable({
write(chunk, _enc, cb) {
buf += chunk.toString();
cb();
},
});
return { stream, text: () => buf };
}
function stripAnsi(s: string): string {
// eslint-disable-next-line no-control-regex
return s.replace(/\x1b\[[0-9;]*[A-Za-z]/g, "");
}
/** Overrides a message's timestamp (the constructor defaults to the current time). */
function at<M extends { timestamp: string }>(ts: string, msg: M): M {
return { ...msg, timestamp: ts };
}
/** token_usage shorthand: request.total = req, session.total = sess (all buckets zero, sufficient for this test group). */
function usage(req: number, sess: number) {
return tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: sess },
{ cache_read: 0, cache_write: 0, output: 0, total: req },
);
}
describe("humanizeTokens", () => {
it("abbreviates with k / M and trims .0", () => {
expect(humanizeTokens(0)).toBe("0");
expect(humanizeTokens(999)).toBe("999");
expect(humanizeTokens(1000)).toBe("1k");
expect(humanizeTokens(1234)).toBe("1.2k");
expect(humanizeTokens(32000)).toBe("32k");
expect(humanizeTokens(1_500_000)).toBe("1.5M");
});
});
describe("pure formatters", () => {
it("formatAbort includes the reason", () => {
expect(stripAnsi(formatAbort({ type: "abort", reason: "ctrl-c" }, t))).toContain("ctrl-c");
});
it("renderHistory includes abort events from resumed sessions", () => {
const { stream, text } = collector();
renderHistory([assistantText("partial", "aborted"), abortEvent("aborted by user")], stream, t);
expect(stripAnsi(text())).toBe("partial [aborted]\n[abort]: aborted by user\n");
});
});
describe("StreamRenderer", () => {
it("streams partial_text deltas and does NOT re-render the complete text", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialText("start", "Hel"));
r.handle(partialText("delta", "lo "));
r.handle(partialText("delta", "world"));
r.handle(partialText("stop", "", "completed"));
r.handle(assistantText("Hello world")); // complete message: must not be re-rendered
expect(stripAnsi(text())).toBe("Hello world\n");
});
it("streams partial_thinking (dim) and skips the complete thinking", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialThinking("start", "think"));
r.handle(partialThinking("delta", "ing"));
r.handle(partialThinking("stop"));
r.handle(thinkingMessage("thinking")); // must not be re-rendered
expect(stripAnsi(text())).toBe("thinking\n");
});
it("does not render a complete tool_call without partials", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c2" }));
expect(text()).toBe("");
});
it("streams partial_tool_call with a pairing tag and skips the complete tool_call", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c4" }));
r.handle(
partialToolCall({ eventType: "delta", name: "", arguments: '{"cmd":"l', toolCallId: "c4" }),
);
r.handle(partialToolCall({ eventType: "delta", name: "", arguments: 's"}', toolCallId: "c4" }));
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c4" }));
r.handle(toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c4" }));
// The call line carries a [tool-<last-3-chars-of-id>] pairing tag matching the output line.
expect(stripAnsi(text())).toBe("[tool-c4] $ ls\n");
});
it("streams partial_tool_call_output with a tagged gutter and skips the complete tool_call_output", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCallOutput({ eventType: "start", toolCallId: "c3" }));
r.handle(partialToolCallOutput({ eventType: "delta", output: "line1\n", toolCallId: "c3" }));
r.handle(partialToolCallOutput({ eventType: "delta", output: "line2", toolCallId: "c3" }));
r.handle(partialToolCallOutput({ eventType: "stop", toolCallId: "c3" }));
r.handle(toolCallOutput({ output: "line1\nline2", toolCallId: "c3" })); // must not be re-rendered
// Each line starts with a tagged gutter (no indent) matching the call line.
expect(stripAnsi(text())).toBe("[tool-c3] >> line1\n[tool-c3] >> line2\n");
});
it("prints the retry line only when the retry request actually begins", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(requestBegin());
r.handle(requestEnd("malformed"));
expect(stripAnsi(text())).toBe(""); // the failure itself prints nothing; only the retry's start does
r.handle(requestBegin()); // retry #1 begins
expect(stripAnsi(text())).toContain("retry #1");
r.handle(requestEnd("timeout"));
r.handle(requestBegin()); // retry #2 begins
expect(stripAnsi(text())).toContain("retry #2");
// Retry #2 fails again and retries are exhausted: no next request_begin, only abort — no retry #3 appears.
r.handle(requestEnd("malformed"));
r.handle(abortEvent("malformed response failed after 2 retries"));
expect(stripAnsi(text())).not.toContain("retry #3");
// The first request of the next run is not a retry, so it prints nothing; a new failure after it counts from 1 again.
r.handle(requestBegin());
r.handle(requestEnd("timeout"));
r.handle(requestBegin());
const lines = stripAnsi(text());
expect(lines.match(/retry #1/g)).toHaveLength(2);
expect(lines).not.toContain("retry #3");
});
it("locks the screen to one streaming tool output; other messages queue until its stop", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCallOutput({ eventType: "start", toolCallId: "tA" }));
r.handle(partialToolCallOutput({ eventType: "delta", output: "a1\n", toolCallId: "tA" }));
// The screen is locked by tA: other streaming messages queue up.
r.handle(partialText("start", ""));
r.handle(partialText("delta", "hello"));
r.handle(partialToolCallOutput({ eventType: "delta", output: "a2\n", toolCallId: "tA" }));
expect(stripAnsi(text())).toBe("[tool-tA] >> a1\n[tool-tA] >> a2\n"); // hello is still queued
r.handle(partialToolCallOutput({ eventType: "stop", toolCallId: "tA" }));
r.handle(partialText("stop", "", "completed"));
expect(stripAnsi(text())).toBe("[tool-tA] >> a1\n[tool-tA] >> a2\nhello\n");
});
it("queues everything while a user prompt is active and flushes after it ends", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.beginUserPrompt();
r.handle(partialText("start", ""));
r.handle(partialText("delta", "after prompt"));
r.handle(partialText("stop", "", "completed"));
expect(text()).toBe(""); // the screen is locked while waiting for user input
r.endUserPrompt();
expect(stripAnsi(text())).toBe("after prompt\n");
});
it("does not print token_usage per turn; endTask prints [stats] line with per-task deltas", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(
sessionMeta({
session_id: "s",
provider: "custom",
model_id: "m",
model_context_window: 1,
system_prompt: "sp",
tools: [{ name: "exec_command", description: "test tool" }],
thinking_level: "medium",
agent_state: "/a",
workspace: "/w",
}),
);
// Two turns: request total 1500, 4000. Per-task token delta = 5500; session cumulative = 12000;
// context = the latest request's input+output (= total) = 4000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 200, total: 1500 },
),
);
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 12000 },
{ cache_read: 0, cache_write: 0, output: 300, total: 4000 },
),
);
expect(stripAnsi(text())).toBe(""); // no stats line is printed mid-turn
r.endTask(2345);
// Exact full-line assertion: context 4k (the latest request's total) and its delta, cumulative tokens 12k,
// per-task delta 5.5k (1500 + 4000), elapsed 2.3s (first task: session equals the delta);
// this also implies session_meta is not rendered (no /w or similar field appears in the output).
expect(stripAnsi(text())).toBe(
"[stats] context 4k (+4k) · tokens 12k (+5.5k) · 2.3s (+2.3s)\n",
);
});
it("accumulates session elapsed across tasks; context delta is vs previous task", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// task 1: context 4000, elapsed 2000ms.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 4000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 4000 },
),
);
r.endTask(2000);
// task 2: context 7000 (+3000 vs. the previous task), session elapsed cumulative 5000ms (this task +3000ms).
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 11000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 7000 },
),
);
r.endTask(3000);
const lines = stripAnsi(text()).trim().split("\n");
const last = lines[lines.length - 1]!;
// Exact full-line assertion: context 7k (delta = 7000 - 4000), cumulative session tokens 11k,
// per-task token delta 7k, total session elapsed 5s (this task +3s).
expect(last).toBe("[stats] context 7k (+3k) · tokens 11k (+7k) · 5s (+3s)");
});
it("context delta goes negative after compaction shrinks the context (no clamping)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// task 1: context 7000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 7000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 7000 },
),
);
r.endTask(1000);
// task 2: context drops to 2000 after compaction -> delta is negative (2000 - 7000 = -5k), not clamped to non-negative.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 9000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 2000 },
),
);
r.endTask(1000);
const lines = stripAnsi(text()).trim().split("\n");
expect(lines[lines.length - 1]).toBe("[stats] context 2k (-5k) · tokens 9k (+2k) · 2s (+1s)");
});
it("renders mode-specific compaction messages (summarize vs discard)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(compactionBegin({ reason: "context", mode: "summarize", context: 150, turns: 3 }));
r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "completed" }));
r.handle(compactionBegin({ reason: "manual", mode: "discard", context: 10, turns: 1 }));
r.handle(compactionEnd({ reason: "manual", mode: "discard", status: "completed" }));
r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "failed" }));
expect(stripAnsi(text())).toBe(
[
"[compaction] summarizing context (context)…",
"[compaction] done; continuing with the summarized context",
"[compaction] discarding context (manual)…",
"[compaction] done; old context discarded",
"[compaction] failed; keeping the current context",
"",
].join("\n"),
);
});
it("轮结束后的压缩:压缩完成行展示本次消耗,但不计入本轮统计增量;不更新上下文", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// Ordinary request: context 5000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 5000 },
),
);
// The compaction request's usage sits between the paired compaction events: no ordinary request_end
// follows it in this turn -> compaction after the turn has ended.
r.handle(compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }));
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 14000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 6000 },
),
);
r.handle(compactionEnd({ reason: "context", mode: "summarize", status: "completed" }));
r.endTask(1000);
const s = stripAnsi(text());
// The compaction-done line still shows this call's usage: session cumulative 14k + this compaction's 6k.
expect(s).toContain(
"[compaction] done; continuing with the summarized context · tokens 14k (+6k)",
);
// Stats line: context stays at the ordinary-request figure of 5k; cumulative tokens 14k (includes
// compaction, following the provider), but this turn's **delta** is only the ordinary request's 5k —
// compaction after the turn ends is not attributed to this turn.
expect(s).toContain("context 5k");
expect(s).toContain("tokens 14k (+5k)");
});
it("轮途中的压缩(其后还有普通 request_end):用时含压缩跨度,Token 增量计入压缩", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// own1: ordinary request, request 5000, 00:00 -> 00:02.
r.handle(at("2026-07-05T00:00:00.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:01.000Z", usage(5000, 5000)));
r.handle(at("2026-07-05T00:00:02.000Z", requestEnd("completed")));
// Mid-turn compaction: 00:03 -> 00:13, request 6000 (the compaction's own summarization request).
r.handle(
at(
"2026-07-05T00:00:03.000Z",
compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }),
),
);
r.handle(at("2026-07-05T00:00:04.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:10.000Z", usage(6000, 14000)));
r.handle(at("2026-07-05T00:00:12.000Z", requestEnd("completed")));
r.handle(
at(
"2026-07-05T00:00:13.000Z",
compactionEnd({ reason: "context", mode: "summarize", status: "completed" }),
),
);
// The turn continues after compaction (carry-over): own2 request 2000, final request_end at 00:16 -> settles the compaction usage.
r.handle(at("2026-07-05T00:00:14.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:15.000Z", usage(2000, 16000)));
r.handle(at("2026-07-05T00:00:16.000Z", requestEnd("completed")));
r.endTask(999); // the passed-in wall clock is ignored: with a request_end present, elapsed comes from the timestamp span
const s = stripAnsi(text());
// Elapsed = first event 00:00 -> the last non-compaction request_end 00:16 = 16s (includes the 10s of
// compaction in the middle, which occupied this turn's wall clock).
// Token delta = own1 5000 + own2 2000 + compaction 6000 = 13k; context uses the ordinary-request figure after compaction, 2k.
expect(s).toContain("context 2k");
expect(s).toContain("tokens 16k (+13k)");
expect(s).toContain("16s (+16s)");
});
it("轮结束后的压缩(带 request 事件):用时止于压缩前的最后一个 request_end", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// own1: 00:00 -> 00:03.
r.handle(at("2026-07-05T00:00:00.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:01.000Z", usage(5000, 5000)));
r.handle(at("2026-07-05T00:00:03.000Z", requestEnd("completed")));
// Trailing compaction: 00:04 -> 00:24, a full 20s, with no ordinary request_end for this turn after it.
r.handle(
at(
"2026-07-05T00:00:04.000Z",
compactionBegin({ reason: "context", mode: "summarize", context: 5000, turns: 1 }),
),
);
r.handle(at("2026-07-05T00:00:05.000Z", requestBegin()));
r.handle(at("2026-07-05T00:00:20.000Z", usage(6000, 14000)));
r.handle(at("2026-07-05T00:00:23.000Z", requestEnd("completed")));
r.handle(
at(
"2026-07-05T00:00:24.000Z",
compactionEnd({ reason: "context", mode: "summarize", status: "completed" }),
),
);
r.endTask(999);
const s = stripAnsi(text());
// Elapsed = 00:00 -> the last non-compaction request_end before compaction, 00:03 = 3s (the whole 20s
// compaction span comes after it and does not count).
// Token delta is only own1's 5k; compaction's 6k is not attributed to this turn.
expect(s).toContain("tokens 14k (+5k)");
expect(s).toContain("3s (+3s)");
});
it("renders approval_decision events (approved / denied)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(approvalDecision("allow", "c1"));
r.handle(approvalDecision("deny", "c2"));
const s = stripAnsi(text());
expect(s).toContain("[approved]");
expect(s).toContain("[denied]");
});
it("keeps call → decision contiguous at prompt time and dedupes the late approval_decision event", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const tc = toolCall({ name: "exec_command", arguments: '{"cmd":"pwd"}', toolCallId: "p8" });
// Interactive approval: while locked, renders "call line -> (prompt, written directly by readline) -> result" as three contiguous lines.
r.beginUserPrompt(tc);
r.noteApprovalDecision(tc, "allow");
r.endUserPrompt();
expect(stripAnsi(text())).toBe("[tool-p8] $ pwd\n✓ [approved]\n");
// A late approval_decision event is deduped by key and not re-rendered.
r.handle(approvalDecision("allow", "p8"));
expect(stripAnsi(text())).toBe("[tool-p8] $ pwd\n✓ [approved]\n");
});
it("re-renders a half-streamed call line at approval and suppresses its late tail deltas", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const tc = toolCall({
name: "exec_command",
arguments: '{"cmd":"git status"}',
toolCallId: "h7",
});
// The call line is still mid-stream (only half its arguments rendered) when approval begins.
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "h7" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"git st',
toolCallId: "h7",
}),
);
r.beginUserPrompt(tc);
// The trailing delta / stop arrive queued while the screen is locked.
r.handle(
partialToolCall({ eventType: "delta", name: "", arguments: 'atus"}', toolCallId: "h7" }),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "h7" }));
r.noteApprovalDecision(tc, "allow");
r.endUserPrompt();
const s = stripAnsi(text());
// At approval time, the full call line is re-rendered in place from the complete message, right next to
// the result; after unlocking, the late tail is deduped and must not start a duplicate call line after
// the result line.
expect(s).toContain("[tool-h7] $ git status\n✓ [approved]\n");
expect(s.slice(s.indexOf("[approved]"))).not.toContain("[tool-h7]");
});
it("defers another call's auto-approval rendering while an interactive prompt is active", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const parent = toolCall({
name: "exec_command",
arguments: '{"cmd":"pwd"}',
toolCallId: "pa1",
});
const child = withOrigin(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "ch2" }),
"sess_kid",
);
r.beginUserPrompt(parent); // parent call's interactive prompt: locks the screen
r.noteApprovalDecision(child, "allow"); // concurrent subagent auto-approval: deferred, not inserted mid-prompt
expect(stripAnsi(text())).not.toContain("ch2");
r.noteApprovalDecision(parent, "allow"); // the prompt owner's result renders in place as usual
r.endUserPrompt();
const s = stripAnsi(text());
// Order: parent call line -> parent result -> child call line -> child result.
const iParentOk = s.indexOf("[approved]");
const iChildCall = s.indexOf("[agent-kid-tool-ch2]");
expect(s.indexOf("[tool-pa1]")).toBeGreaterThanOrEqual(0);
expect(iChildCall).toBeGreaterThan(iParentOk);
expect(s.indexOf("[approved]", iChildCall)).toBeGreaterThan(iChildCall);
});
it("endCompact settles manual /compact usage so the next task's delta excludes it", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 5000 },
),
);
r.endTask(1000);
// Manual /compact: the compaction request consumes 6000 (already shown on the compaction-done line), endCompact settles it.
r.handle(compactionBegin({ reason: "manual", mode: "summarize", context: 5000, turns: 1 }));
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 14000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 6000 },
),
);
r.handle(compactionEnd({ reason: "manual", mode: "summarize", status: "completed" }));
r.endCompact(500);
// The next task consumes only 1000: its delta must not include compaction's 6000.
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 15000 },
{ cache_read: 0, cache_write: 0, output: 0, total: 1000 },
),
);
r.endTask(1000);
const lines = stripAnsi(text()).trim().split("\n");
expect(lines[lines.length - 1]).toContain("tokens 15k (+1k)");
});
it("re-renders the call line next to the decision when other output separated them (auto-approve)", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// The call line is first rendered while streaming, then separated from the decision by other output.
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c5" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"ls"}',
toolCallId: "c5",
}),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c5" }));
r.handle(partialText("start", ""));
r.handle(partialText("delta", "hi"));
r.handle(partialText("stop", "", "completed"));
// Auto-approval: the call line is no longer adjacent -> it is re-rendered in place, with the result immediately following it as a pair.
r.noteApprovalDecision(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c5" }),
"allow",
);
expect(stripAnsi(text())).toBe("[tool-c5] $ ls\nhi\n[tool-c5] $ ls\n✓ [approved]\n");
});
it("does not re-render the call line when it is already adjacent to the decision", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "c6" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"ls"}',
toolCallId: "c6",
}),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "c6" }));
r.noteApprovalDecision(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "c6" }),
"deny",
);
expect(stripAnsi(text())).toBe("[tool-c6] $ ls\n× [denied]\n");
});
});
describe("StreamRenderer — nested (origin-tagged) subagent messages", () => {
const hop: MessageOrigin = "sess_child";
it("renders nested tool calls with an agent-tool tag; skips nested text/thinking partials", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// Nested text/thinking is not rendered (the child's reply is shown via the parent tool's output gutter).
r.handle(withOrigin(partialText("delta", "child text"), hop));
r.handle(withOrigin(partialThinking("delta", "child think"), hop));
// A nested complete tool_call renders one line (so the user can see what tool the subagent is calling
// before approval); the tag is agent-<last-3-chars-of-child-session>-tool-<last-3-chars-of-id>; the
// approval line carries no tag.
r.handle(
withOrigin(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "cc1" }),
hop,
),
);
r.handle(withOrigin(approvalDecision("allow", "cc1"), hop));
expect(stripAnsi(text())).toBe("[agent-ild-tool-cc1] $ ls\n✓ [approved]\n");
});
it("renders the pending nested tool call at approval time when its stream copy has not arrived; dedupes the late copy", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
const tc = withOrigin(
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "cc9" }),
hop,
);
// The approval callback arrives before the forwarded message: beginUserPrompt renders the call line directly from the complete message.
r.beginUserPrompt(tc);
expect(stripAnsi(text())).toBe("[agent-ild-tool-cc9] $ ls\n");
r.endUserPrompt();
// The late forwarded copy is deduped by key and not re-rendered.
r.handle(tc);
expect(stripAnsi(text())).toBe("[agent-ild-tool-cc9] $ ls\n");
});
it("renders the pending parent tool call at approval time and suppresses its late partial stream", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.beginUserPrompt(
toolCall({ name: "exec_command", arguments: '{"cmd":"pwd"}', toolCallId: "p7" }),
);
r.endUserPrompt();
// The whole late streaming copy is deduped and skipped.
r.handle(partialToolCall({ eventType: "start", name: "exec_command", toolCallId: "p7" }));
r.handle(
partialToolCall({
eventType: "delta",
name: "",
arguments: '{"cmd":"pwd"}',
toolCallId: "p7",
}),
);
r.handle(partialToolCall({ eventType: "stop", name: "", toolCallId: "p7" }));
expect(stripAnsi(text())).toBe("[tool-p7] $ pwd\n");
});
it("adds nested token_usage request totals to the task delta and the session total", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
// One parent-session request: 1500; one child-session request: 2000 -> per-task delta 3.5k;
// session cumulative = parent 8000 + child 2000 = 10k (delta and cumulative use the same basis: parent + child).
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 8000 },
{ cache_read: 0, cache_write: 0, output: 200, total: 1500 },
),
);
r.handle(
withOrigin(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 2000 },
{ cache_read: 0, cache_write: 0, output: 100, total: 2000 },
),
hop,
),
);
r.endTask(1000);
const s1 = stripAnsi(text());
expect(s1).toContain("3.5k"); // the per-task delta includes child-session usage
expect(s1).toContain("10k"); // the session cumulative includes child-session usage
// The child session's cumulative persists across tasks: the next task consumes only from the parent session, cumulative = 9000 + 2000 = 11k (+1k).
r.handle(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 9000 },
{ cache_read: 0, cache_write: 0, output: 100, total: 1000 },
),
);
r.endTask(1000);
const lines = stripAnsi(text()).trim().split("\n");
const last = lines[lines.length - 1]!;
expect(last).toContain("11k");
expect(last).toContain("+1k");
});
it("prints stats when a task only has nested (subagent) token usage", () => {
const { stream, text } = collector();
const r = new StreamRenderer(stream, t);
r.handle(
withOrigin(
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 2000 },
{ cache_read: 0, cache_write: 0, output: 100, total: 2000 },
),
hop,
),
);
r.endTask(1000);
const s = stripAnsi(text());
expect(s).toContain("[stats]");
expect(s).toContain("2k (+2k)");
});
});
describe("renderHistory (resume)", () => {
it("renders complete messages statically with interruption markers", async () => {
const { renderHistory } = await import("../src/render.js");
const { userText } = await import("@prismshadow/penguin-core");
const { stream, text } = collector();
renderHistory(
[
userText("hello"),
thinkingMessage("pondering"),
assistantText("hi there"),
toolCall({ name: "exec_command", arguments: '{"cmd":"ls"}', toolCallId: "call_653" }),
toolCallOutput({ output: "a.txt\nb.txt", toolCallId: "call_653" }),
assistantText("half answer", "aborted"),
],
stream,
);
const s = stripAnsi(text());
expect(s).toContain("> hello");
expect(s).toContain("pondering");
expect(s).toContain("hi there");
expect(s).toContain("[tool-653] $ ls");
expect(s).toContain("[tool-653] >> a.txt");
expect(s).toContain("[tool-653] >> b.txt");
// An interrupted message carries a marker (rendering includes the interrupted turn).
expect(s).toContain("half answer [aborted]");
});
it("skips events and renders nothing for empty history", async () => {
const { renderHistory } = await import("../src/render.js");
const { stream, text } = collector();
renderHistory(
[
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: 1 },
{ cache_read: 0, cache_write: 0, output: 0, total: 1 },
),
],
stream,
);
expect(text()).toBe("");
});
});
+72
View File
@@ -0,0 +1,72 @@
import { describe, expect, it } from "vitest";
import { Command } from "commander";
import {
DEFAULT_HOST,
DEFAULT_PORT,
browserCommand,
browserUrl,
registerServeCommands,
resolvePort,
} from "../src/commands/serve.js";
import { getMessages } from "../src/i18n.js";
describe("resolvePort(选项 > 环境变量 > 缺省 7364)", () => {
it("都未给时用缺省 7364", () => {
expect(DEFAULT_PORT).toBe(7364);
expect(resolvePort(undefined, undefined)).toBe(7364);
expect(resolvePort(undefined, "")).toBe(7364); // an empty string counts as unset
});
it("只有环境变量时取环境变量", () => {
expect(resolvePort(undefined, "8080")).toBe(8080);
});
it("选项优先于环境变量", () => {
expect(resolvePort("9000", "8080")).toBe(9000);
});
it("非法值(非整数 / 越界)抛错", () => {
expect(() => resolvePort("abc", undefined)).toThrow(/abc/);
expect(() => resolvePort("3.14", undefined)).toThrow();
expect(() => resolvePort("-1", undefined)).toThrow();
expect(() => resolvePort("65536", undefined)).toThrow();
expect(() => resolvePort(undefined, "not-a-port")).toThrow(/not-a-port/);
});
});
describe("browserCommand(按平台选择打开命令)", () => {
const url = "http://127.0.0.1:7364/";
it("darwin → open", () => {
expect(browserCommand("darwin", url)).toEqual({ command: "open", args: [url] });
});
it("win32 → cmd /c start(空标题占位在 URL 前)", () => {
expect(browserCommand("win32", url)).toEqual({
command: "cmd",
args: ["/c", "start", "", url],
});
});
it("其他平台(linux 等)→ xdg-open", () => {
expect(browserCommand("linux", url)).toEqual({ command: "xdg-open", args: [url] });
expect(browserCommand("freebsd", url)).toEqual({ command: "xdg-open", args: [url] });
});
});
describe("browserUrl(通配监听地址转 127.0.0.1)", () => {
it("常规 host 原样拼接", () => {
expect(browserUrl(DEFAULT_HOST, 7364)).toBe("http://127.0.0.1:7364/");
expect(browserUrl("192.168.1.2", 8080)).toBe("http://192.168.1.2:8080/");
});
it("0.0.0.0 / :: 时浏览器 URL 用 127.0.0.1", () => {
expect(browserUrl("0.0.0.0", 7364)).toBe("http://127.0.0.1:7364/");
expect(browserUrl("::", 7364)).toBe("http://127.0.0.1:7364/");
});
});
describe("registerServeCommands(命令注册)", () => {
it("注册 server 与 web 两个顶层命令,web 缺省 open=true(--no-open 可关)", () => {
const program = new Command();
registerServeCommands(program, getMessages("en"));
const names = program.commands.map((c) => c.name());
expect(names).toContain("server");
expect(names).toContain("web");
const web = program.commands.find((c) => c.name() === "web")!;
expect(web.opts().open).toBe(true);
});
});
+51
View File
@@ -0,0 +1,51 @@
/**
* runTask's result reporting: when a Task ends with a main-session abort event (LLM
* failure / reconnect exhausted / user interrupt), it reports aborted=true, which
* `penguin run` maps to a non-zero exit code; a sub-session abort does not count.
*/
import { describe, expect, it } from "vitest";
import { Writable } from "node:stream";
import { abortEvent, assistantText, withOrigin } from "@prismshadow/penguin-core";
import type { OmniMessage, Session } from "@prismshadow/penguin-core";
import { StreamRenderer } from "../src/render.js";
import { runTask } from "../src/task-loop.js";
import { getMessages } from "../src/i18n.js";
const t = getMessages("en");
function fakeSession(messages: OmniMessage[]): Session {
return {
async *run() {
for (const m of messages) yield m;
},
toolPermission: () => "rw",
} as unknown as Session;
}
function silentRenderer(): StreamRenderer {
const stream = new Writable({
write(_chunk, _enc, cb) {
cb();
},
});
return new StreamRenderer(stream, t);
}
describe("runTask abort reporting", () => {
it("reports aborted=true when the task ends with a main-session abort event", async () => {
const result = await runTask(fakeSession([abortEvent("llm request error: 401")]), [], {
renderer: silentRenderer(),
t,
});
expect(result.aborted).toBe(true);
});
it("reports aborted=false on normal completion; child-session aborts do not count", async () => {
const result = await runTask(
fakeSession([withOrigin(abortEvent("child aborted"), "sess_child"), assistantText("done")]),
[],
{ renderer: silentRenderer(), t },
);
expect(result.aborted).toBe(false);
});
});
+89
View File
@@ -0,0 +1,89 @@
import { describe, expect, it } from "vitest";
import { renderPartialToolCall } from "../src/tool-render.js";
describe("renderPartialToolCall", () => {
it("renders partial exec_command args as $ <cmd-so-far>", () => {
expect(renderPartialToolCall("exec_command", '{"cmd":')).toBeNull();
expect(renderPartialToolCall("exec_command", '{"cmd":"l')).toBe("$ l");
expect(renderPartialToolCall("exec_command", '{"cmd":"ls"}')).toBe("$ ls");
expect(renderPartialToolCall("exec_command", '{"cmd":"echo \\"hi\\"')).toBe('$ echo "hi"');
});
it("renders run_subagent as run_subagent << <prompt>, folded to one line", () => {
expect(renderPartialToolCall("run_subagent", '{"prompt":')).toBeNull();
expect(renderPartialToolCall("run_subagent", '{"prompt":"analy')).toBe("run_subagent << analy");
expect(renderPartialToolCall("run_subagent", '{"prompt":"line1\\nline2"}')).toBe(
"run_subagent << line1 line2",
);
});
it("renders input_command polls (empty chars) without a payload", () => {
expect(renderPartialToolCall("input_command", '{"process_id":')).toBeNull();
expect(renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d"}')).toBe(
"⌨ input_command → proc-1a2b3c4d",
);
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":""}'),
).toBe("⌨ input_command → proc-1a2b3c4d");
});
it("renders non-empty input_command chars with visible control characters", () => {
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"y\\n"}'),
).toBe("⌨ input_command → proc-1a2b3c4d << y\\n");
// U+0003 (Ctrl-C) is rendered in caret notation.
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"\\u0003"}'),
).toBe("⌨ input_command → proc-1a2b3c4d << ^C");
// Disambiguates literal backslash escapes: chars "a", "\", "n" render as a\\n, distinct from a real newline \n.
expect(
renderPartialToolCall("input_command", '{"process_id":"proc-1a2b3c4d","chars":"a\\\\n"}'),
).toBe("⌨ input_command → proc-1a2b3c4d << a\\\\n");
});
it("keeps input_command previews append-only across \\uXXXX delta boundaries", () => {
const stages = [
'{"process_id":"proc-1a2b3c4d","chars":"y',
'{"process_id":"proc-1a2b3c4d","chars":"y\\u0',
'{"process_id":"proc-1a2b3c4d","chars":"y\\u0003',
];
const previews = stages.map((s) => renderPartialToolCall("input_command", s)!);
expect(previews[0]).toBe("⌨ input_command → proc-1a2b3c4d << y");
// An incomplete \u escape is treated as "stop here" rather than emitting the raw hex as literal text.
expect(previews[1]).toBe("⌨ input_command → proc-1a2b3c4d << y");
expect(previews[2]).toBe("⌨ input_command → proc-1a2b3c4d << y^C");
for (let i = 1; i < previews.length; i++) {
expect(previews[i]!.startsWith(previews[i - 1]!)).toBe(true);
}
});
it("renders input_subagent polls without a payload and follow-up prompts with one", () => {
expect(
renderPartialToolCall("input_subagent", '{"subagent_id":"subagent-9f8e7d6c","prompt":""}'),
).toBe("⌨ input_subagent → subagent-9f8e7d6c");
expect(
renderPartialToolCall(
"input_subagent",
'{"subagent_id":"subagent-9f8e7d6c","prompt":"continue with the tests"}',
),
).toBe("⌨ input_subagent → subagent-9f8e7d6c << continue with the tests");
});
it("truncates long payload previews and stops growing afterwards", () => {
const long = "x".repeat(130);
const capped = renderPartialToolCall(
"input_subagent",
`{"subagent_id":"subagent-9f8e7d6c","prompt":"${long}"}`,
);
expect(capped).toBe(`⌨ input_subagent → subagent-9f8e7d6c << ${"x".repeat(120)}…`);
const longer = renderPartialToolCall(
"input_subagent",
`{"subagent_id":"subagent-9f8e7d6c","prompt":"${long}yyy"}`,
);
expect(longer).toBe(capped);
});
it("falls back to name(args-prefix) for unknown tools", () => {
expect(renderPartialToolCall("search", '{"q":"hi')).toBe('search({"q":"hi');
});
});
+7
View File
@@ -0,0 +1,7 @@
{
"extends": "../../tsconfig.base.json",
"compilerOptions": {
"rootDir": "."
},
"include": ["src", "test"]
}
+22
View File
@@ -0,0 +1,22 @@
import { defineConfig } from "tsup";
export default defineConfig({
entry: ["src/index.ts"],
format: ["esm"],
target: "node20",
platform: "node",
clean: true,
sourcemap: true,
// Bundle the workspace source @prismshadow/penguin-core, but keep third-party deps (incl.
// CJS yaml / smol-toml / agenthub) external and resolved from node_modules at runtime —
// avoids bundling CJS deps into ESM and triggering a "Dynamic require" error.
// @prismshadow/penguin-skills must stay external: it reads files under its own skills/ dir
// at runtime (files are the source of truth); bundling would break paths relative to the
// package root. cli already declares it as a direct dependency.
// @prismshadow/penguin-server stays external: the penguin server / web commands import it
// dynamically at runtime.
// tsup treats this package's package.json dependencies as external by default, so these
// deps are already declared there.
noExternal: ["@prismshadow/penguin-core"],
banner: { js: "#!/usr/bin/env node" },
});
+41
View File
@@ -0,0 +1,41 @@
# @prismshadow/penguin-core
The PenguinHarness SDK and execution engine: the ReAct loop (`context_engine`), the OmniMessage protocol, the LLM / Environment interface contracts, Agent State and append-only Traces.
The engine speaks only OmniMessage and delegates everything else through two swappable interfaces — `LLMInterface` (models, via the [`@prismshadow/agenthub`](https://www.npmjs.com/package/@prismshadow/agenthub) gateway) and `EnvironmentInterface` (tool execution). The SDK caller is the Human boundary: one entry point, `session.run`, streams the whole loop.
```ts
import { createAgent, isCompleteModelMessage, userText } from "@prismshadow/penguin-core";
const agent = await createAgent({ agentId: "default_agent" });
const session = await agent.createSession({ workspaceDir: process.cwd() });
for await (const output of session.run([userText("Create hello.txt containing hi")], {
approve: async () => "allow", // per-tool-call approval
})) {
if (isCompleteModelMessage(output) && output.payload.type === "text") {
console.log(output.payload.text);
}
}
```
A single `run` drives a complete Task: streaming output, per-call approvals, concurrent tool execution, interrupt carry-over, automatic reconnect and context compaction. State lives under `~/.penguin/data` (`PENGUIN_HOME`); every Session restores fully from its Trace.
## Documentation
- [Architecture](https://prism-shadow.github.io/penguin-harness/docs/architecture)
- [The OmniMessage Protocol](https://prism-shadow.github.io/penguin-harness/docs/omni-message)
- [Core Interfaces](https://prism-shadow.github.io/penguin-harness/docs/interfaces)
- [The Agent Loop](https://prism-shadow.github.io/penguin-harness/docs/agent-loop)
- [Sessions & Traces](https://prism-shadow.github.io/penguin-harness/docs/sessions-and-traces)
## Development
```bash
pnpm --filter @prismshadow/penguin-core build # tsup → dist/ (exports point at dist)
pnpm --filter @prismshadow/penguin-core typecheck
pnpm --filter @prismshadow/penguin-core test
pnpm test:e2e # live-model e2e (needs DEEPSEEK_API_KEY)
```
Part of [PenguinHarness](https://github.com/Prism-Shadow/penguin-harness) · Apache-2.0
+58
View File
@@ -0,0 +1,58 @@
{
"name": "@prismshadow/penguin-core",
"version": "0.0.1",
"type": "module",
"description": "PenguinHarness core SDK: context_engine, OmniMessage protocol, LLM/Environment interfaces.",
"license": "Apache-2.0",
"repository": {
"type": "git",
"url": "git+https://github.com/Prism-Shadow/penguin-harness.git",
"directory": "packages/core"
},
"exports": {
".": {
"types": "./dist/index.d.ts",
"import": "./dist/index.js"
},
"./omnimessage": {
"types": "./dist/omnimessage/index.d.ts",
"import": "./dist/omnimessage/index.js"
},
"./interfaces": {
"types": "./dist/interfaces.d.ts",
"import": "./dist/interfaces.js"
},
"./model-catalog": {
"types": "./dist/state/model-catalog.d.ts",
"import": "./dist/state/model-catalog.js"
}
},
"main": "./dist/index.js",
"types": "./dist/index.d.ts",
"scripts": {
"typecheck": "tsc --noEmit -p tsconfig.json",
"test": "vitest run --passWithNoTests",
"test:e2e": "PENGUIN_E2E=1 vitest run test/llm.e2e.test.ts",
"build": "tsup"
},
"dependencies": {
"@prismshadow/agenthub": "^0.3.3",
"@prismshadow/penguin-skills": "workspace:*",
"smol-toml": "^1.3.0",
"yaml": "^2.5.0"
},
"devDependencies": {
"@types/node": "^24.0.0",
"dotenv": "^17.0.0",
"tsup": "^8.3.0",
"typescript": "^5.6.0",
"vitest": "^2.1.0"
},
"files": [
"dist",
"LICENSE"
],
"publishConfig": {
"access": "public"
}
}
+643
View File
@@ -0,0 +1,643 @@
/**
* Agent and the `createAgent` entry point.
*
* `createAgent` is the unified way to create/load an Agent: it initializes Agent State
* if the directory is empty, otherwise loads by agentId.
* An Agent has exactly one Agent State and can run multiple times; the
* Workspace is determined when a Session is created.
*/
import fs from "node:fs/promises";
import path from "node:path";
import {
assertValidId,
assembleSystemPrompt,
buildToolConfig,
selectBuiltinToolsForModel,
DEFAULT_COMPACTION_PROMPT,
formatModelRef,
getModel,
listInstalledSkills,
loadAgentVault,
loadOrInitAgentState,
loadProjectConfig,
projectDir,
resolveModelRef,
scratchpadDir,
systemConfigPath,
tracesDir,
type AgentState,
type ModelRef,
type ProjectConfig,
} from "./state/index.js";
import { GenerativeModel, ToolCallIdAllocator } from "./llm/index.js";
import { Environment } from "./environment/index.js";
import {
Writer,
findLatestTraceFile,
latestSessionId as latestTraceSessionId,
readTraceTolerant,
resumeTrace,
} from "./trace/index.js";
import { Session } from "./session.js";
import {
createTempWorkspace,
formatSessionId,
sessionEnvironment,
} from "./internal/session-support.js";
import { userText, withOrigin } from "./omnimessage/index.js";
import type {
MessageOrigin,
OmniMessage,
TokenCounts,
ToolCallPayload,
} from "./omnimessage/index.js";
import { SUBAGENT_NAME } from "./environment/tools/run-subagent.js";
import { INPUT_SUBAGENT_NAME } from "./environment/tools/input-subagent.js";
import type { CompactionSettings } from "./engine/context-engine.js";
import type {
GenerativeModelConfig,
SubagentRunner,
ToolDefinition,
VisionDescriberService,
} from "./interfaces.js";
import type { ModelEntry } from "./state/index.js";
/**
* Maximum subagent spawn depth. Currently capped at 1 level (a subagent cannot spawn
* another subagent); the depth mechanism is designed to support multiple levels —
* raise this constant to allow deeper nesting.
*/
const MAX_SUBAGENT_DEPTH = 1;
export interface CreateAgentOptions {
agentId?: string;
projectId?: string;
/** Local data root directory; defaults to `resolveRoot()` (PENGUIN_HOME or ~/.penguin/data). */
root?: string;
}
export interface CreateSessionOptions {
/** Workspace for this run; if unspecified, a temporary Workspace is created under the Agent directory. */
workspaceDir?: string;
/** Model used for this Session (upstream model_id); if unspecified, uses the Project's default Model. */
modelId?: string;
/**
* Provider grouping for `modelId` (a paired reference); if omitted, resolved via
* `resolveModelRef` semantics — `model_id` only resolves if it is a globally unique
* exact match in the config; zero or multiple matches produce a clear error.
*/
provider?: string;
/** Explicit credentials; if unspecified, falls back to credentials in the Project config, then to AgentHub reading environment variables. */
apiKey?: string;
baseUrl?: string;
/** Internal use: this Session's depth in the subagent spawn chain (0 at the top level), used to cap spawn depth. */
subagentDepth?: number;
}
export interface ResumeSessionOptions {
/** Id of the Session to resume. */
sessionId: string;
/** Explicit credentials; if unspecified, falls back to credentials in the Project config, then to AgentHub reading environment variables. */
apiKey?: string;
baseUrl?: string;
}
/**
* Effective compaction threshold: capped at 75% of the model's `context_window` —
* the threshold must stay well below the hard window limit, otherwise small-window
* models get rejected by the provider (a non-retryable 400) before compaction even
* triggers, and the compaction request itself (old context + prompt + summary output)
* also needs headroom. Not clamped when `<=0` (disabled) or the window is unknown.
*/
export function effectiveMaxContextLength(configured: number, contextWindow: unknown): number {
if (configured <= 0) return configured;
if (typeof contextWindow !== "number") return configured;
return Math.min(configured, Math.floor(contextWindow * 0.75));
}
/** Create or load an Agent. */
export async function createAgent(opts: CreateAgentOptions = {}): Promise<Agent> {
const state = await loadOrInitAgentState(opts);
const projectConfig = await loadProjectConfig(state.root, state.projectId);
return new Agent(state, projectConfig);
}
export class Agent {
constructor(
readonly state: AgentState,
readonly projectConfig: ProjectConfig,
) {}
/**
* Create a Session in the specified (or a temporary) Workspace.
* Docs: /docs/sessions-and-traces § "Run model".
*/
async createSession(opts: CreateSessionOptions = {}): Promise<Session> {
// Model is validated first (before creating the Workspace, so failure leaves no
// temp directory behind): the reference must resolve to an entry in the Project
// config (the (provider, model_id) pair is the unique key); a reference
// outside the config throws immediately rather than passing silently — otherwise
// credentials, pricing, and the context window would all be unavailable.
if (opts.modelId === undefined && opts.provider !== undefined) {
throw new Error(
"指定了 provider 却未指定 modelId:模型引用须成对给出(provider 不能单独使用)。",
);
}
let ref: ModelRef;
if (opts.modelId !== undefined) {
// The only entry point for resolving an "omitted provider" reference (resolveModelRef): three branches — unique match / zero matches / ambiguous.
ref = resolveModelRef(this.projectConfig, opts.modelId, opts.provider);
} else if (this.projectConfig.default_model) {
ref = this.projectConfig.default_model;
} else {
throw new Error(
"未指定 modelId,且 Project 配置中没有 default_model。请用 `penguin config model add/default` 设置默认模型。",
);
}
const modelEntry = getModel(this.projectConfig, ref);
if (!modelEntry) {
throw new Error(
`Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`,
);
}
// Credentials are inlined on the model entry (single config file); an
// explicit argument takes priority, falling back to AgentHub reading env vars
// when both are absent.
const apiKey = opts.apiKey ?? modelEntry.api_key;
const baseUrl = opts.baseUrl ?? modelEntry.base_url;
// An explicit Workspace must already exist as a directory: if it
// doesn't, throw rather than auto-create (to avoid a typo silently working in
// the wrong location); a temp Workspace is only created when unspecified.
let workspaceDir: string;
if (opts.workspaceDir) {
workspaceDir = path.resolve(opts.workspaceDir);
let stat;
try {
stat = await fs.stat(workspaceDir);
} catch {
throw new Error(
`Workspace 不存在:${workspaceDir}。请指定一个已存在的目录,或不指定 Workspace 以使用临时目录。`,
);
}
if (!stat.isDirectory()) {
throw new Error(`Workspace 不是目录:${workspaceDir}。`);
}
} else {
workspaceDir = await createTempWorkspace(
this.state.root,
this.state.projectId,
this.state.agentId,
);
}
const sessionId = formatSessionId();
const subagentDepth = opts.subagentDepth ?? 0;
// Agent-level vault (agent_state/.vault.toml) and installed Skills: read the current values each time a Session is created.
const vault = await loadAgentVault(this.state.root, this.state.projectId, this.state.agentId);
const installedSkills = await listInstalledSkills(
this.state.root,
this.state.projectId,
this.state.agentId,
);
// The assembled system prompt goes both to the LLM and into session_meta (so the
// Trace can audit the actual effective value). The vault only injects **key names**
// into the prompt (so the model knows which API keys are available); values only
// go into the subprocess environment. Skills only inject metadata (name and
// description); the model reads the body on demand via shell.
const systemPrompt = assembleSystemPrompt(
this.state,
sessionEnvironment(workspaceDir, sessionId, {
agentId: this.state.agentId,
projectDir: projectDir(this.state.root, this.state.projectId),
}),
Object.keys(vault),
installedSkills,
);
const rt = await this.buildRuntime({
workspaceDir,
modelEntry,
apiKey,
baseUrl,
systemPrompt,
subagentDepth,
vault,
});
const trace = new Writer({
tracesDir: tracesDir(this.state.root, this.state.projectId, this.state.agentId),
sessionId,
});
return new Session({
meta: {
session_id: sessionId,
provider: modelEntry.provider,
model_id: modelEntry.model_id,
model_context_window: modelEntry.context_window ?? "unknown",
system_prompt: systemPrompt,
tools: rt.tools,
thinking_level: this.state.systemConfig.model?.thinking_level ?? "default",
agent_state: this.state.stateDir,
workspace: workspaceDir,
},
llm: rt.llm,
environment: rt.environment,
trace,
createLLM: rt.createLLM,
createBareLLM: rt.createBareLLM,
compaction: rt.compaction,
// Model doesn't support images: input images are written to the session scratchpad and their paths appended to the text (viewed via describe_image).
...(modelEntry.vision === false
? {
inputImagesDir: path.join(
scratchpadDir(this.state.root, this.state.projectId, this.state.agentId),
sessionId,
),
}
: {}),
// Max turns comes from the Agent's system_config (runtime parameters belong to the Agent config).
...(this.state.systemConfig.max_turns !== undefined
? { maxTurns: this.state.systemConfig.max_turns }
: {}),
});
}
/**
* Resume an existing Session and continue the conversation.
*
* The resume source is the Session's **latest-index** Trace file: runtime config is
* read from its `session_meta` (Model, the original system prompt text, and the
* Workspace all carry over from the original Session and cannot be changed), while
* tools and Environment are reassembled from the current Agent State. The replayed,
* already-committed history is injected once via AgentHub's setHistory (used only on
* resume); any leftover input is rebuilt as carry-over (paired fallback placeholders
* are synthesized in memory only, never written to the Trace). Messages after resume
* continue in the original Trace file (the file follows the context, not the date),
* and Token / turn-count stats carry over from their original values.
* Docs: /docs/sessions-and-traces § "Session recovery".
*/
async resumeSession(opts: ResumeSessionOptions): Promise<Session> {
const { sessionId } = opts;
const dir = tracesDir(this.state.root, this.state.projectId, this.state.agentId);
const located = await findLatestTraceFile(dir, sessionId);
if (!located) {
throw new Error(`Session 不存在:${sessionId}(在 ${dir} 下未找到对应 Trace 文件)。`);
}
const resumed = resumeTrace(await readTraceTolerant(located.path));
if (!resumed.meta) {
throw new Error(`Trace 缺少 session_meta,无法恢复:${located.path}`);
}
const meta = resumed.meta.payload;
// Model reference is stored as a pair in session_meta; a missing provider means legacy data (no migration since the product hasn't shipped yet).
if (typeof meta.provider !== "string") {
throw new Error(
`Trace 来自旧版本数据(session_meta 缺少 provider,模型引用未分列):${located.path}。请删除数据目录后重建会话。`,
);
}
// The Workspace carries over from the original Session and must still exist (throw if missing, never auto-create).
const workspaceDir = meta.workspace;
let stat;
try {
stat = await fs.stat(workspaceDir);
} catch {
throw new Error(`原 Session 的 Workspace 已不存在:${workspaceDir},无法恢复。`);
}
if (!stat.isDirectory()) {
throw new Error(`原 Session 的 Workspace 不是目录:${workspaceDir},无法恢复。`);
}
// The Model carries over from the original Session (paired reference) and must still be present in the Project config.
const ref: ModelRef = { provider: meta.provider, model_id: meta.model_id };
const modelEntry = getModel(this.projectConfig, ref);
if (!modelEntry) {
throw new Error(
`原 Session 的 Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model add\` 重新配置后再恢复。`,
);
}
const apiKey = opts.apiKey ?? modelEntry.api_key;
const baseUrl = opts.baseUrl ?? modelEntry.base_url;
// Tools and Environment are reassembled from the current Agent State (tool
// definitions are passed with every Request and aren't part of the history); the
// system prompt uses the original text recorded in the Trace (identical to the
// original history); the vault uses current values (it's injected into the
// subprocess environment, not the history, so a resumed Session should get the
// latest keys too).
const rt = await this.buildRuntime({
workspaceDir,
modelEntry,
apiKey,
baseUrl,
systemPrompt: meta.system_prompt,
subagentDepth: 0,
vault: await loadAgentVault(this.state.root, this.state.projectId, this.state.agentId),
});
// History is injected once into a fresh context object (setHistory is only used
// on resume); Session cumulative Token counts carry over. Wrap the error
// descriptively: bad tool arguments in the history (e.g. truncated JSON written by
// a third-party OpenAI-compatible endpoint) throw a raw SyntaxError during
// conversion, so the error must indicate Trace history corruption rather than a
// regular runtime error.
if (resumed.history.length > 0) {
try {
rt.llm.setHistory(resumed.history);
} catch (err) {
const detail = err instanceof Error ? err.message : String(err);
throw new Error(
`恢复失败:Trace 历史无法注入(记录可能损坏,如非法的工具参数 JSON):${detail}`,
);
}
}
rt.llm.sessionTokens = resumed.sessionTokens;
// Continue writing to the original Trace file (the Trace only records real messages; synthesized paired placeholders are re-emitted in memory alongside carry-over).
const trace = new Writer({
tracesDir: dir,
sessionId,
dateDir: located.dateDir,
startIndex: located.index,
});
return new Session({
meta: {
session_id: sessionId,
provider: modelEntry.provider,
model_id: modelEntry.model_id,
model_context_window: modelEntry.context_window ?? "unknown",
system_prompt: meta.system_prompt,
tools: rt.tools,
thinking_level: this.state.systemConfig.model?.thinking_level ?? "default",
agent_state: this.state.stateDir,
workspace: workspaceDir,
},
llm: rt.llm,
environment: rt.environment,
trace,
createLLM: rt.createLLM,
createBareLLM: rt.createBareLLM,
compaction: rt.compaction,
// Model doesn't support images: input images are written to the session scratchpad and their paths appended to the text (viewed via describe_image).
...(modelEntry.vision === false
? {
inputImagesDir: path.join(
scratchpadDir(this.state.root, this.state.projectId, this.state.agentId),
sessionId,
),
}
: {}),
...(this.state.systemConfig.max_turns !== undefined
? { maxTurns: this.state.systemConfig.max_turns }
: {}),
// session_meta is already in the original Trace file, so it isn't rewritten; on the first write after a compaction-triggered rotation, the file is split first.
metaAlreadyWritten: true,
initialEngineState: {
carryOver: resumed.carryOver,
...(resumed.pendingSummary ? { pendingSummary: resumed.pendingSummary } : {}),
sessionTurns: resumed.sessionTurns,
sessionTokens: resumed.sessionTokens,
lastRequestTotal: resumed.lastRequestTotal,
pendingTraceRotation: resumed.contextClosed,
},
resumedHistory: resumed.renderMessages,
});
}
/** Id of the most recent Session under the current Agent (determined by the timestamp in session_id); returns null if there is no Session. */
async latestSessionId(): Promise<string | null> {
return latestTraceSessionId(
tracesDir(this.state.root, this.state.projectId, this.state.agentId),
);
}
/**
* Assemble a Session's runtime components (shared by createSession and
* resumeSession): the child-Agent runner, Environment and tools, the LLM object
* and its post-compaction rebuild factory, and the compaction config.
*/
private async buildRuntime(args: {
workspaceDir: string;
/** This Session's Model entry: the caller (createSession / resumeSession) has already validated it exists in the config. */
modelEntry: ModelEntry;
apiKey: string | undefined;
baseUrl: string | undefined;
systemPrompt: string;
subagentDepth: number;
vault: Record<string, string>;
}): Promise<{
environment: Environment;
tools: ToolDefinition[];
llm: GenerativeModel;
createLLM: (sessionTokens: TokenCounts) => GenerativeModel;
createBareLLM: () => GenerativeModel;
compaction: CompactionSettings;
}> {
const { workspaceDir, modelEntry, apiKey, baseUrl, systemPrompt, subagentDepth, vault } = args;
// Child-Agent runner: injected into the run_subagent tool so it doesn't need to
// depend on Agent/Session (breaking a circular dependency). The model can
// optionally choose agentId (omitted = call the current Agent) and modelId
// (omitted = Project default). Precheck errors (depth limit exceeded / agent
// doesn't exist) are expressed as throws, which the Environment collapses to failed.
// Docs: /docs/interfaces § "Subagent interfaces"
const parentAgent = this;
const { root, projectId, agentId: parentAgentId } = this.state;
const subagentRunner: SubagentRunner = {
// Spawn and run are separate: the same child Session can run for multiple turns
// (continuing via input_subagent appending a prompt); resource cleanup is
// consolidated in handle.dispose (called by the managing ManagedSubagentSession).
async spawn({ agentId, modelId }) {
if (subagentDepth >= MAX_SUBAGENT_DEPTH) {
throw new Error(
`subagent depth limit ${MAX_SUBAGENT_DEPTH} reached; not spawning another subagent`,
);
}
if (agentId !== undefined && agentId !== parentAgentId) {
try {
assertValidId("agent_id", agentId);
await fs.access(systemConfigPath(root, projectId, agentId));
} catch {
throw new Error(
`subagent error: agent "${agentId}" does not exist or is not accessible`,
);
}
}
const childAgent =
agentId !== undefined && agentId !== parentAgentId
? await createAgent({ root, projectId, agentId })
: parentAgent;
const childSession = await childAgent.createSession({
workspaceDir,
...(modelId !== undefined ? { modelId } : {}),
subagentDepth: subagentDepth + 1,
});
// All child-session messages are tagged with an origin (the child Session id,
// prepended as one hop from outer to inner); the first turn forwards the
// child's session_meta first (including agent_state and other metadata) so the
// parent frontend can recognize the nested session (for rendering, stats,
// approval visibility); the parent Trace skips these accordingly (the child
// Session has its own Trace, linked by session id).
const hop: MessageOrigin = childSession.sessionId;
let metaSent = false;
return {
sessionId: hop,
async *run({ prompt, signal, approve }) {
if (!metaSent) {
metaSent = true;
yield withOrigin(childSession.metaMessage, hop);
}
// Pass through the parent's approval callback: the child Session inherits
// the parent Agent's approval mode (with no callback, the child engine
// defaults to deny). The tool_call received for approval also carries the
// origin, so the approval UI can identify which tool a subagent is calling.
const childApprove = approve
? (tc: OmniMessage<ToolCallPayload>) => approve(withOrigin(tc, hop))
: undefined;
for await (const msg of childSession.run([userText(prompt)], {
...(signal ? { signal } : {}),
...(childApprove ? { approve: childApprove } : {}),
})) {
yield withOrigin(msg, hop);
}
},
dispose() {
childSession.dispose();
},
};
},
};
// Tool exposure is capped by depth: a (leaf) child Agent that has reached the
// max spawn depth no longer gets run_subagent or input_subagent (the latter
// depends on the subagent_id produced by the former, so exposing it alone is
// meaningless).
const canSpawn = subagentDepth < MAX_SUBAGENT_DEPTH;
const baseToolConfig = buildToolConfig(this.state);
// Select tool entries by the session model's type (marked via forModel: vision
// models use read_image, text-only models use describe_image; entries without
// this marker are unaffected).
const modelVision = modelEntry.vision !== false;
let customTools = selectBuiltinToolsForModel(baseToolConfig.customTools, modelVision);
if (!canSpawn) {
customTools = customTools.filter(
(d) => d.name !== SUBAGENT_NAME && d.name !== INPUT_SUBAGENT_NAME,
);
}
const toolConfig = { ...baseToolConfig, customTools };
// When the session model doesn't support images (vision=false): inject a vision
// model service for describe_image (forModel: "text-only", selected by the filter
// above) — images are described by the Project config's vision_model (a paired
// reference), and the tool returns text. Even when unconfigured or invalid, it is
// still injected (modelId=null); the tool then finishes with a failed explanation,
// and images are never allowed into that session's history.
let visionDescriber: VisionDescriberService | undefined;
if (modelEntry.vision === false) {
const visionRef = this.projectConfig.vision_model;
const visionEntry = visionRef ? getModel(this.projectConfig, visionRef) : undefined;
if (visionEntry && visionEntry.vision !== false) {
visionDescriber = {
// The model attribution in the tool output matches the request's source: both are the entry's upstream model_id.
modelId: visionEntry.model_id,
createLLM: () =>
new GenerativeModel({
modelId: visionEntry.model_id,
...(visionEntry.api_key !== undefined ? { apiKey: visionEntry.api_key } : {}),
...(visionEntry.base_url !== undefined ? { baseUrl: visionEntry.base_url } : {}),
...(visionEntry.client_type !== undefined
? { clientType: visionEntry.client_type }
: {}),
tools: [],
thinkingLevel: "none",
maxTokens: 2048,
requestTimeoutMs: 60_000,
}),
};
} else {
visionDescriber = { modelId: null };
}
}
// Environment binds the Workspace and tool config; tools are listed first so
// GenerativeModel can be initialized. Vault environment variables are injected
// into command subprocesses (shared by createSession and resumeSession; the
// caller reads the current agent_state/.vault.toml); a child Agent loads **its
// own** vault via createAgent rather than inheriting the parent's.
const environment = new Environment({
workspaceDir,
toolConfig,
services: { subagentRunner, ...(visionDescriber ? { visionDescriber } : {}) },
...(Object.keys(vault).length > 0 ? { vault } : {}),
});
const tools = await environment.listTools();
// LLM constructor args are extracted into a constant so they can be reused as-is when
// rebuilding a new LLM object after compaction (with a fresh model context) — the system
// prompt and tool definitions aren't part of the compacted history, so the new object keeps
// them unchanged. The model id sent to AgentHub is always the entry's upstream `model_id`
// (client_type inference/passing follows it); session_meta, Trace, usage, pricing, and catalog
// matching all use the (provider, model_id) pair as the primary key.
// The tool_call_id uniqueness registry is shared with the new LLM rebuilt from llmConfig after
// compaction: its uniqueness scope is the Session's whole render span, so same-named tool calls
// after compaction don't collide with earlier tool cards' ids.
const llmConfig: GenerativeModelConfig = {
modelId: modelEntry.model_id,
toolCallIds: new ToolCallIdAllocator(),
...(apiKey !== undefined ? { apiKey } : {}),
...(baseUrl !== undefined ? { baseUrl } : {}),
...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}),
tools,
systemPrompt,
...(modelEntry.context_window !== undefined
? { contextWindow: modelEntry.context_window }
: {}),
...(this.state.systemConfig.model?.max_tokens !== undefined
? { maxTokens: this.state.systemConfig.model.max_tokens }
: {}),
...(this.state.systemConfig.model?.thinking_level !== undefined
? { thinkingLevel: this.state.systemConfig.model.thinking_level }
: {}),
...(this.state.systemConfig.model?.timeoutMs !== undefined
? { requestTimeoutMs: this.state.systemConfig.model.timeoutMs }
: {}),
};
const llm = new GenerativeModel(llmConfig);
const createLLM = (sessionTokens: TokenCounts): GenerativeModel => {
const next = new GenerativeModel(llmConfig);
// Carries over the Session's cumulative Token counts, so token_usage.session stays continuous across compaction.
next.sessionTokens = sessionTokens;
return next;
};
// Bare LLM for one-off out-of-band requests (meta requests like generateTitle):
// same Model/credentials, no tools, no system prompt, thinking disabled, a small
// output cap, and an independent timeout.
const createBareLLM = (): GenerativeModel =>
new GenerativeModel({
modelId: modelEntry.model_id,
...(apiKey !== undefined ? { apiKey } : {}),
...(baseUrl !== undefined ? { baseUrl } : {}),
...(modelEntry.client_type !== undefined ? { clientType: modelEntry.client_type } : {}),
tools: [],
thinkingLevel: "none",
maxTokens: 300,
requestTimeoutMs: 30_000,
});
// Compaction config: defaults are filled in here; an unknown mode falls back to summarize (the default).
const compactionConfig = this.state.systemConfig.compaction;
const compaction: CompactionSettings = {
maxContextLength: effectiveMaxContextLength(
compactionConfig?.max_context_length ?? 128000,
modelEntry.context_window,
),
maxSessionTurns: compactionConfig?.max_session_turns ?? -1,
mode: compactionConfig?.mode === "discard" ? "discard" : "summarize",
prompt: compactionConfig?.prompt ?? DEFAULT_COMPACTION_PROMPT,
};
return { environment, tools, llm, createLLM, createBareLLM, compaction };
}
}
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,380 @@
/**
* Environment —— executes approved tool calls inside the Workspace.
*
* Environment has no knowledge of any specific tool: it only assembles the tool names supported
* by ToolConfig into BuiltinTool instances (see `environment/tools/`), and dispatches execution
* by looking up the tool name. Adding a new built-in tool only requires implementing BuiltinTool
* and registering it — no changes to this file needed. Tool call **rendering** is not core's
* concern; it's handled by the CLI / Web frontend.
*
* The **framing and finalization** of the tool stream is handled uniformly by Environment:
* - Entering execution immediately emits `start`; the tool only needs to yield output deltas
* (its own start/stop are ignored);
* - Output is truncated online **front-to-back** by maxOutputLength (head is kept, forwarding
* stops once exceeded); the truncation marker, the tool's self-reported end marker
* (`ToolResult.note`, e.g. exit code — appended outside the truncation, never lost even when
* long output is truncated), and timeout/interruption/error markers are all emitted as part
* of the stream — **the content produced by concatenating streamed chunks matches the full
* message exactly**;
* - Nested session messages carrying an origin marker (e.g. forwarded from run_subagent) pass
* through unchanged, taking no part in this tool's output or finalization;
* - Argument parsing failures, unknown tool names, tool throws, and other exceptions all
* collapse into an explanatory, complete `tool_call_output` — never throws — and **output is
* never empty under any circumstance**.
* Docs: /docs/tools § "Execution contract".
*/
import { partialToolCallOutput, toolCallOutput } from "../omnimessage/index.js";
import type { OmniMessage, StopReason } from "../omnimessage/index.js";
import type {
EnvironmentConfig,
EnvironmentInterface,
ToolConfig,
ToolDefinition,
ToolExecutionRequest,
ToolPermission,
} from "../interfaces.js";
import type { BuiltinTool, ToolResult } from "./tools/types.js";
import { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js";
import { CommandSessionManager } from "./tools/command/index.js";
import { SubagentSessionManager } from "./tools/subagent/index.js";
/** Default cap on tool output truncation (characters). */
const DEFAULT_MAX_OUTPUT_LENGTH = 16000;
/** Default timeout cap for a single tool call (milliseconds); <=0 disables it (every tool must be bound by timeoutMs). */
const DEFAULT_TOOL_TIMEOUT_MS = 120000;
/** Marker appended to the result when a tool is interrupted by the user. */
const TOOL_ABORTED_NOTE = "[interrupted: tool aborted by user]";
/** Placeholder marker used when a tool produces no output at all (tool_call_output content is never empty). */
const TOOL_EMPTY_NOTE = "[no output]";
/**
* Explanation for a failed argument JSON parse. The normal pipeline never reaches this: bad
* JSON already throws during AgentHub's parsing stage, and the LLM layer finalizes it as
* malformed for the engine to reconnect (see generative-model.ts) — it's never dispatched into
* Environment as a completed tool_call. This function is only a defensive fallback for the
* public interface.
*/
function describeArgumentsError(name: string, raw: string, err: unknown): string {
const detail = err instanceof Error ? err.message : String(err);
if (raw.trim() === "") {
return `Tool call "${name}" failed: the arguments field is empty. Re-issue the call with a complete JSON object.`;
}
return `Tool call "${name}" failed: the arguments are not valid JSON (${detail}). Re-issue the call with one complete, valid JSON object.`;
}
/** Appends a marker after existing content: newline-joins if content is non-empty, otherwise just returns the marker. */
function appendNote(base: string, note: string): string {
return base ? `${base}\n${note}` : note;
}
/** The delta needed to stream out `note` on top of existing content `base` (includes separator, same basis as appendNote). */
function noteSuffix(base: string, note: string): string {
return base ? `\n${note}` : note;
}
export class Environment implements EnvironmentInterface {
private readonly workspaceDir: string;
private readonly toolConfig: ToolConfig;
/** Assembled built-in tools: tool name -> BuiltinTool. Only tools supported by the registry and present in config. */
private readonly tools: Map<string, BuiltinTool>;
/** Long-running command session registry: constructed within this Environment and shared between exec_command / input_command. */
private readonly commandSessions: CommandSessionManager;
/** Background subagent session registry: constructed within this Environment and shared between run_subagent / input_subagent. */
private readonly subagentSessions: SubagentSessionManager;
constructor(config: EnvironmentConfig) {
this.workspaceDir = config.workspaceDir;
this.toolConfig = config.toolConfig;
this.tools = new Map();
// The background session registry is created alongside Environment (one per Session) and
// injected into whichever tools need it; all sessions are finalized together on dispose.
// The vault environment variables are injected into child processes by the command session
// registry at spawn time.
this.commandSessions = new CommandSessionManager(
config.vault !== undefined ? { vault: config.vault } : {},
);
this.subagentSessions = new SubagentSessionManager();
const services = {
...config.services,
commandSessions: this.commandSessions,
subagentSessions: this.subagentSessions,
};
// Assemble the tools supported by config into BuiltinTool instances; unrecognized tool
// names are skipped (neither exposed to the LLM nor executable).
for (const def of config.toolConfig.customTools) {
const factory = BUILTIN_TOOL_FACTORIES[def.name];
if (factory) this.tools.set(def.name, factory(def, services));
}
}
/** Releases runtime resources held by Environment: finalizes all managed background sessions (command and subagent). Idempotent. */
dispose(): void {
this.commandSessions.dispose();
this.subagentSessions.dispose();
}
/**
* Lists tools available to the current Session, for context_engine to initialize GenerativeModel.
* Only lists tools that have been assembled (i.e. supported by the registry) — tool names
* unrecognized in config are not exposed to the LLM (consistent with the constructor);
* the definition (description/parameters) treats **the config entry as the single source of
* truth** — factories must not rewrite the definition at runtime; where a differentiated
* implementation is needed, use a separate explicit tool-name entry with a `forModel`
* annotation (e.g. read_image / describe_image).
* Only exposes `{name, description, parameters}`, dropping permission/maxOutputLength.
* MCP Server config flows into Environment via toolConfig; enumerating concrete MCP tools
* is left to a later adapter layer.
*/
async listTools(): Promise<ToolDefinition[]> {
return this.toolConfig.customTools
.filter((tool) => this.tools.has(tool.name))
.map((tool) => ({
name: tool.name,
description: tool.description,
...(tool.parameters !== undefined ? { parameters: tool.parameters } : {}),
}));
}
/** Looks up a tool's permission level (for the frontend's permission-mode decisions); returns undefined for an unknown tool. */
toolPermission(name: string): ToolPermission | undefined {
return this.toolConfig.customTools.find((t) => t.name === name)?.permission;
}
/**
* Executes an approved tool call, streaming `partial_tool_call_output` and a final
* `tool_call_output`; nested messages carrying origin pass through unchanged. Dispatches by
* looking up the tool name; any exception collapses into an explanatory output — never throws.
*
* The priority for deciding stop_reason is: user interruption > timeout > tool throw > tool
* self-report. Interruption is determined by the `signal` held by Environment, and is
* compatible with both a tool self-reporting aborted and an AbortError raised by the
* interruption. An internal abort raised by a timeout does not count as a user interruption —
* it's finalized as failed, with the timeout reason written into the output.
* Docs: /docs/tools § "Execution contract".
*/
async *executeTool(request: ToolExecutionRequest): AsyncGenerator<OmniMessage> {
const payload = request.toolCall.payload;
// tool_call_id is passed through unchanged, so context_engine and the LLM can associate the
// request with its result.
const toolCallId = payload.tool_call_id;
const name = payload.name;
// Every path is framed uniformly by Environment: entering execution emits start; the end
// uniformly emits stop + the full message.
yield partialToolCallOutput({ eventType: "start", toolCallId });
const tool = this.tools.get(name);
if (!tool) {
yield* emitFailure(toolCallId, `Unknown tool: ${name}`);
return;
}
// Parse the tool's argument JSON; a parse failure collapses into an explanatory output
// (also streamed, so the frontend can render it).
let parsed: unknown;
try {
parsed = JSON.parse(payload.arguments);
} catch (err) {
yield* emitFailure(toolCallId, describeArgumentsError(name, payload.arguments, err));
return;
}
const args =
parsed !== null && typeof parsed === "object" ? (parsed as Record<string, unknown>) : {};
const maxOutputLength = tool.definition.maxOutputLength ?? DEFAULT_MAX_OUTPUT_LENGTH;
const timeoutMs = tool.definition.timeoutMs ?? DEFAULT_TOOL_TIMEOUT_MS;
const signal = request.signal;
// User interruption and tool timeout are merged into a single internal signal handed to the
// tool: either one triggers abortion of execution.
// The timeout constraint is enforced uniformly by Environment for all tools; the
// tool only needs to respond to signal.
const ac = new AbortController();
if (signal?.aborted) ac.abort();
const onAbort = (): void => ac.abort();
signal?.addEventListener("abort", onAbort, { once: true });
let timedOut = false;
const timer =
timeoutMs > 0
? setTimeout(() => {
timedOut = true;
ac.abort();
}, timeoutMs)
: null;
timer?.unref?.();
// Consume the tool stream: content deltas are forwarded after online front-truncation;
// nested messages pass through; manual iteration to capture the generator's return value.
let streamed = ""; // Content forwarded so far (<= maxOutputLength)
let contentLen = 0; // Total length of content produced by the tool (including truncated/discarded parts)
let toolOutput: string | null = null; // Fallback: content basis when the tool produces a full message itself
let selfReported: StopReason | undefined; // Tool's self-reported stop reason (return value takes priority over the full message)
let selfNote: string | null = null; // Tool's self-reported end marker (e.g. exit code), appended outside truncation
let selfImages: string[] | undefined; // Tool's self-reported images (data URL), carried via a single streamed delta and the full message
let thrown: unknown = null;
const gen = tool.execute(args, {
workspaceDir: this.workspaceDir,
toolCallId,
signal: ac.signal,
// Pass through the parent's approve callback (run_subagent uses it so the child Session
// inherits the parent's approval mode; other tools ignore it).
...(request.approve ? { approve: request.approve } : {}),
});
try {
for (;;) {
const res = await gen.next();
if (res.done) {
const result: ToolResult | void = res.value;
if (result?.stopReason) selfReported = result.stopReason;
if (result?.note) selfNote = result.note;
if (result?.images && result.images.length > 0) selfImages = result.images;
break;
}
const out = res.value;
if (out.origin && out.origin.length > 0) {
yield out; // Nested session message: pass through unchanged, not part of this tool's output/finalization
continue;
}
const p = out.payload as {
type?: string;
event_type?: string;
stop_reason?: string;
output?: string;
};
if (p.type === "partial_tool_call_output") {
// Only takes delta content; start/stop are ignored (framing is uniformly handled by Environment).
if (p.event_type !== "delta" || !p.output) continue;
contentLen += p.output.length;
// maxOutputLength <= 0 means truncation is disabled (same semantics as timeoutMs).
const room =
maxOutputLength > 0 ? maxOutputLength - streamed.length : Number.POSITIVE_INFINITY;
if (room > 0) {
const chunk = p.output.length > room ? p.output.slice(0, room) : p.output;
streamed += chunk;
// Rebuild the delta: tool_call_id is uniformly enforced by Environment, never trusting the tool's own value.
yield partialToolCallOutput({
eventType: "delta",
output: chunk,
toolCallId,
});
}
} else if (p.type === "tool_call_output") {
// Fallback: if the tool still produces a full message, use it as the basis for content and stop reason (not needed under the new contract).
toolOutput = p.output ?? "";
if (selfReported === undefined && p.stop_reason) {
selfReported = p.stop_reason as StopReason;
}
} else {
// Other message types without origin: protocol misuse, ignore and warn (keep the parent stream clean).
process.stderr.write(
`[penguin] tool "${name}" yielded unexpected message type "${p.type}"; ignored.\n`,
);
}
}
} catch (err) {
// A tool throw also collapses into the uniform finalization: keep already-streamed content, don't discard produced output.
thrown = err;
} finally {
if (timer) clearTimeout(timer);
signal?.removeEventListener("abort", onAbort);
}
// Uniform finalization. Content basis = the tool's self-produced full message (fallback
// path) or the already-forwarded delta; after front-truncating to the cap, the truncation
// marker and interruption/timeout/error markers are appended in turn, all made up via
// streamed deltas — streamed concatenation == the full message.
const contentBase = toolOutput ?? streamed;
const capped =
maxOutputLength > 0 && contentBase.length > maxOutputLength
? contentBase.slice(0, maxOutputLength)
: contentBase;
const truncated = capped.length < contentBase.length || contentLen > streamed.length;
const aborted =
signal?.aborted === true ||
(!timedOut &&
(selfReported === "aborted" ||
(thrown as { name?: string } | null)?.name === "AbortError"));
let stopReason: StopReason;
const notes: string[] = [];
if (truncated) {
notes.push(`[output truncated: exceeded ${maxOutputLength} chars]`);
}
// The tool's self-reported end marker (e.g. exit code): appended outside the truncation —
// if treated as a content delta it would get cut off once long output hits the cap, and the
// model would misread a command that failed after printing lots of output as successful.
if (selfNote) {
notes.push(selfNote);
}
if (aborted) {
stopReason = "aborted";
notes.push(TOOL_ABORTED_NOTE);
} else if (timedOut) {
stopReason = "failed";
notes.push(`[tool timeout: exceeded ${timeoutMs}ms]`);
} else if (thrown != null) {
stopReason = "failed";
notes.push(`[tool error] ${thrown instanceof Error ? thrown.message : String(thrown)}`);
} else {
stopReason = selfReported ?? "completed";
}
// The tool's reply must never be empty: an empty tool_result leaves the model unable to
// tell "silent success" apart from "call failed", and some Providers outright reject empty
// content blocks.
if (capped === "" && notes.length === 0) {
notes.push(TOOL_EMPTY_NOTE);
}
const noteText = notes.join("\n");
const fullOutput = noteText ? appendNote(capped, noteText) : capped;
// Compensating content delta: if nothing was streamed, emit the whole thing at once; on the
// fallback path, emit the portion of the full message beyond the already-streamed prefix
// (if the tool is internally inconsistent, the full message wins — no further reconciliation).
let compensation = "";
if (streamed === "") compensation = capped;
else if (toolOutput !== null && capped.startsWith(streamed)) {
compensation = capped.slice(streamed.length);
}
if (compensation) {
yield partialToolCallOutput({
eventType: "delta",
output: compensation,
toolCallId,
});
}
if (noteText) {
yield partialToolCallOutput({
eventType: "delta",
output: noteSuffix(capped, noteText),
toolCallId,
});
}
// Images are made up via streaming: images are not delta'd — a single delta carries them
// all at once right before stop, and the full message carries them again — satisfying
// "streamed concatenation == full message" the same way text does (truncation only applies
// to text, never touches images).
// Only carried on normal completion; interruption/timeout/error paths carry no images, to keep finalization simple.
const images = stopReason === "completed" ? selfImages : undefined;
if (images) {
yield partialToolCallOutput({ eventType: "delta", toolCallId, images });
}
yield partialToolCallOutput({ eventType: "stop", toolCallId, stopReason });
yield toolCallOutput({
output: fullOutput,
toolCallId,
stopReason,
...(images ? { images } : {}),
});
}
}
/** Upfront failure (unknown tool/argument parse failure): delta(explanation) -> stop -> full failed output (start already emitted by the caller). */
function* emitFailure(toolCallId: string, message: string): Generator<OmniMessage> {
yield partialToolCallOutput({ eventType: "delta", output: message, toolCallId });
yield partialToolCallOutput({ eventType: "stop", toolCallId, stopReason: "failed" });
yield toolCallOutput({ output: message, toolCallId, stopReason: "failed" });
}
+14
View File
@@ -0,0 +1,14 @@
/**
* Environment module barrel — exports the environment interface implementation and builtin tool abstractions.
*/
export { Environment } from "./environment.js";
export type { BuiltinTool, ToolExecutionContext } from "./tools/types.js";
export { BUILTIN_TOOL_FACTORIES } from "./tools/registry.js";
export type { BuiltinToolFactory } from "./tools/registry.js";
export { createExecCommandTool, EXEC_COMMAND_NAME } from "./tools/exec-command.js";
export { createInputCommandTool, INPUT_COMMAND_NAME } from "./tools/input-command.js";
export { createSubagentTool, SUBAGENT_NAME } from "./tools/run-subagent.js";
export { createInputSubagentTool, INPUT_SUBAGENT_NAME } from "./tools/input-subagent.js";
export { CommandSessionManager, ManagedSession } from "./tools/command/index.js";
export type { ProcessExit, SpawnOptions } from "./tools/command/index.js";
export { SubagentSessionManager, ManagedSubagentSession } from "./tools/subagent/index.js";
@@ -0,0 +1,44 @@
/**
* CappedTextBuffer — capacity-capped unread-text buffer shared by background sessions.
*
* When over capacity, drops the oldest content (keeping the tail) and tallies the dropped count;
* `drain()` prefixes a marker noting the drop count when taking all unread content (guards
* against a chatty background process / sub-agent blowing up memory, see each session class's
* capacity constant).
*/
export class CappedTextBuffer {
private text = "";
private omitted = 0; // Characters dropped due to the capacity cap, not yet read
/** `dropLabel` is used in the drop-marker text, e.g. "earlier output" / "earlier subagent output". */
constructor(
private readonly cap: number,
private readonly dropLabel: string,
) {}
get isEmpty(): boolean {
return this.text.length === 0 && this.omitted === 0;
}
append(chunk: string): void {
this.text += chunk;
if (this.text.length > this.cap) {
const drop = this.text.length - this.cap;
this.text = this.text.slice(drop); // Keep the newest (tail), drop the oldest
this.omitted += drop;
}
}
/** Takes the current unread content (including the drop marker); clears the buffer. */
drain(): string {
if (this.isEmpty) return "";
const b = this.text;
this.text = "";
if (this.omitted > 0) {
const n = this.omitted;
this.omitted = 0;
return `[... ${n} chars of ${this.dropLabel} dropped ...]\n${b}`;
}
return b;
}
}
@@ -0,0 +1,8 @@
/**
* Barrel for shared background-session infrastructure.
*/
export { BackgroundRegistry } from "./registry.js";
export type { BackgroundTask } from "./registry.js";
export { clampYield } from "./limits.js";
export { WakeSignal } from "./wake-signal.js";
export { CappedTextBuffer } from "./capped-buffer.js";
@@ -0,0 +1,23 @@
/**
* Yield-time clamping shared by background-session tools.
*
* `yield_time_ms` is a soft budget for a single tool call: "wait at most until the session ends
* or this duration expires" (expiry yields, it's not a failure). Only a lower bound is set: a
* wait that's too short isn't meaningful and just adds round trips. The upper bound is no longer
* an independent constant — it's derived from the tool's own `timeoutMs` (with a reserved
* margin, so the yield happens before the Environment's timeout fallback fires); no upper bound
* is set when `timeoutMs <= 0` (disabled).
*/
/** Lower bound for yield time (ms). */
export const MIN_YIELD_MS = 250;
/** Margin (ms) reserved between the yield upper bound and the tool's `timeoutMs`: the yield must happen before the timeout fallback. */
const TIMEOUT_MARGIN_MS = 1_000;
/** Clamps the raw argument to `[MIN_YIELD_MS, timeoutMs - margin]`; falls back to `fallback` if not a number, no upper bound when `timeoutMs <= 0`. */
export function clampYield(raw: unknown, fallback: number, timeoutMs?: number): number {
const n = typeof raw === "number" && Number.isFinite(raw) ? raw : fallback;
const lower = Math.max(n, MIN_YIELD_MS);
if (timeoutMs === undefined || timeoutMs <= 0) return lower;
return Math.min(lower, Math.max(timeoutMs - TIMEOUT_MARGIN_MS, MIN_YIELD_MS));
}
@@ -0,0 +1,174 @@
/**
* BackgroundRegistry —— generic registry for background sessions, shared by command sessions
* and subagent sessions.
*
* Responsibilities: allocating and managing background session ids, enforcing the concurrency
* cap, reclaiming idle sessions, uniform finalization when the Session/Environment ends, and a
* hard kill on process 'exit' as a fallback (JS has no destructors, so cleanup must be explicit).
* Background sessions living for days at a time are a legitimate form; idle reclamation is only
* a leak fallback — sessions unaccessed for longer than `IDLE_TTL_MS` are finalized by a
* periodic sweep.
*
* The two concurrency-cap strategies are expressed by how `makeRoom` is called:
* - Command sessions: if full at registration time, prefer evicting exited sessions, otherwise
* evict LRU (killing a background process is an acceptable cost);
* - Subagent sessions: `makeRoom` is called before launch (only evicting completed, idle ones);
* if there's no room, spawning is rejected — evicting a running subagent is equivalent to
* discarding in-progress work, which is semantically unacceptable.
*
* No lock is needed under the single-threaded event loop; but note the registry may change
* across an `await`, so check before using an entry.
* Docs: /docs/tools § "Background session caps".
*/
import { randomUUID } from "node:crypto";
/** Idle reclamation TTL (milliseconds): a session unaccessed for longer than this is treated as a leak and reclaimed. */
const IDLE_TTL_MS = 10 * 24 * 60 * 60_000; // 10 days
/** Idle reclamation check interval (milliseconds): TTL is measured in days, so an hourly sweep is sufficient. */
const IDLE_SWEEP_MS = 60 * 60_000;
/** Minimal contract a background session must satisfy to be managed by the registry. */
export interface BackgroundTask {
/** Timestamp of the most recent access (used for LRU eviction); refreshed by the registry on register/get. */
lastUsed: number;
/** Whether the session is still running (determines eviction priority). */
running: boolean;
/** Asynchronous finalization (SIGTERM -> SIGKILL / abort); idempotent. */
kill(): void;
/** Synchronous hard kill (process 'exit' fallback path: the event loop has stopped, timers are unavailable). */
killHard(): void;
}
// process 'exit' fallback: use a single module-level listener to manage all registries, avoiding
// each Session adding its own listener and triggering EventEmitter's MaxListeners warning.
const LIVE_REGISTRIES = new Set<BackgroundRegistry<BackgroundTask>>();
let exitHookInstalled = false;
function ensureExitHook(): void {
if (exitHookInstalled) return;
exitHookInstalled = true;
process.on("exit", () => {
for (const r of LIVE_REGISTRIES) r.killAllHard();
});
}
export class BackgroundRegistry<T extends BackgroundTask> {
private readonly tasks = new Map<string, T>();
private readonly idPrefix: string;
private readonly maxTasks: number;
private readonly reapTimer: ReturnType<typeof setInterval>;
private disposed = false;
constructor(opts: { idPrefix: string; maxTasks: number }) {
this.idPrefix = opts.idPrefix;
this.maxTasks = opts.maxTasks;
LIVE_REGISTRIES.add(this as unknown as BackgroundRegistry<BackgroundTask>);
ensureExitHook();
this.reapTimer = setInterval(() => this.reapIdle(), IDLE_SWEEP_MS);
this.reapTimer.unref?.();
}
get size(): number {
return this.tasks.size;
}
/**
* Makes room for a new session. Returns true immediately if not full; when full, evicts per
* `evictRunning`:
* - false (subagent): only evicts the least-recently-used **completed** session; if all are
* running, returns false (the caller rejects spawning);
* - true (command): evicts exited sessions first, otherwise LRU-evicts a running one.
*/
makeRoom(evictRunning: boolean): boolean {
if (this.tasks.size < this.maxTasks) return true;
let lruId: string | null = null;
let lruUsed = Number.POSITIVE_INFINITY;
for (const [id, t] of this.tasks) {
if (!t.running) {
this.remove(id); // Prefer evicting sessions that have already ended
return true;
}
if (t.lastUsed < lruUsed) {
lruUsed = t.lastUsed;
lruId = id;
}
}
if (!evictRunning) return false;
if (lruId) this.remove(lruId);
return this.tasks.size < this.maxTasks;
}
/**
* Registers a session, allocating and returning a unique id (`<prefix>-xxxxxxxx`). The caller
* must first free up room via `makeRoom`. `preferredSuffix` is the preferred id suffix (e.g.
* the tail of a child Session id, so the tool handle correlates with the message origin/
* frontend nesting label); falls back to random if omitted or on collision.
*/
register(task: T, preferredSuffix?: string): string {
this.ensureActive();
let id = preferredSuffix ? `${this.idPrefix}-${preferredSuffix}` : this.randomId();
while (this.tasks.has(id)) id = this.randomId();
task.lastUsed = Date.now();
this.tasks.set(id, task);
return id;
}
private randomId(): string {
return `${this.idPrefix}-${randomUUID().replace(/-/g, "").slice(0, 8)}`;
}
/** Looks up a session by id and refreshes its access time; returns undefined if not found. */
get(id: string): T | undefined {
if (this.disposed) return undefined;
const t = this.tasks.get(id);
if (t) t.lastUsed = Date.now();
return t;
}
/** Removes a session from the registry and finalizes it. */
remove(id: string): void {
const t = this.tasks.get(id);
if (!t) return;
this.tasks.delete(id);
t.kill();
}
/** Kills and clears all sessions (called when the Session/Environment ends). */
killAll(): void {
for (const t of this.tasks.values()) t.kill();
this.tasks.clear();
}
/** Synchronously hard-kills all sessions (process 'exit' fallback path). */
killAllHard(): void {
for (const t of this.tasks.values()) t.killHard();
this.tasks.clear();
}
/** Disposes: removes the fallback registration and kills all sessions. Idempotent. */
dispose(): void {
if (this.disposed) return;
this.disposed = true;
clearInterval(this.reapTimer);
LIVE_REGISTRIES.delete(this as unknown as BackgroundRegistry<BackgroundTask>);
this.killAll();
}
/** Whether the registry has been disposed (the host Session has ended). */
get isDisposed(): boolean {
return this.disposed;
}
/** Reclaims sessions idle for longer than `IDLE_TTL_MS` (leak fallback, triggered by the periodic sweep). */
private reapIdle(): void {
const cutoff = Date.now() - IDLE_TTL_MS;
for (const [id, t] of this.tasks) {
if (t.lastUsed <= cutoff) this.remove(id);
}
}
private ensureActive(): void {
if (this.disposed) {
throw new Error("background session registry disposed");
}
}
}
@@ -0,0 +1,43 @@
/**
* WakeSignal —— a single wakeup point shared by background sessions.
*
* Producer events (data arrival / run finished / new approval request) call `notify()`;
* waiters use `wait(ms)` to wait for "woken up" or expiry, whichever comes first. `notify`
* swaps in a new promise before resolving the old one, so a waiter that wakes up just
* re-checks state — it never misses an event that immediately follows.
*/
export class WakeSignal {
private promise!: Promise<void>;
private resolve!: () => void;
constructor() {
this.arm();
}
private arm(): void {
this.promise = new Promise<void>((resolve) => {
this.resolve = resolve;
});
}
/** Wakes up all waiters: swaps in a new promise before resolving the old one (avoids missing an event that immediately follows). */
notify(): void {
const r = this.resolve;
this.arm();
r();
}
/** Waits for "woken up" or `ms` to elapse, whichever comes first. */
async wait(ms: number): Promise<void> {
let timer: ReturnType<typeof setTimeout> | null = null;
const timeout = new Promise<void>((resolve) => {
// wait is part of an active operation: the timer needs to keep the process alive, and is cleaned up immediately below after notify.
timer = setTimeout(resolve, ms);
});
try {
await Promise.race([this.promise, timeout]);
} finally {
if (timer) clearTimeout(timer);
}
}
}
@@ -0,0 +1,11 @@
/**
* Barrel for the long-running command session module.
*/
export { CommandSessionManager } from "./session-manager.js";
export { ManagedSession, resultForExit } from "./session.js";
export type { ProcessExit, SpawnOptions } from "./session.js";
export {
DEFAULT_EXEC_YIELD_MS,
DEFAULT_WRITE_YIELD_MS,
DEFAULT_EMPTY_POLL_YIELD_MS,
} from "./limits.js";
@@ -0,0 +1,15 @@
/**
* Default yield durations for long-running command sessions.
*
* `yield_time_ms` is the soft budget for a tool call to "wait at most until the command ends or
* this duration elapses" (yielding on expiry is not a failure); see `../background/limits.ts`
* for the clamping logic: it only sets a floor, the ceiling is derived from the tool's own
* `timeoutMs`.
*/
/** Default wait duration (milliseconds) for `exec_command` starting a command. */
export const DEFAULT_EXEC_YIELD_MS = 60_000;
/** Default wait duration (milliseconds) for `input_command` when there's a write. */
export const DEFAULT_WRITE_YIELD_MS = 250;
/** Default wait duration (milliseconds) for `input_command` on an empty poll. */
export const DEFAULT_EMPTY_POLL_YIELD_MS = 5_000;
@@ -0,0 +1,83 @@
/**
* CommandSessionManager — registry and lifecycle management for long-running command sessions.
*
* Constructed by Environment (one per Session), injected via services and shared by the
* `exec_command` and `input_command` tools. Registry responsibilities (id allocation, concurrency
* cap, dispose, process 'exit' fallback) are handled by the generic `BackgroundRegistry` (shared
* with subagent sessions, see `../background/registry.ts`); this class only retains
* command-domain logic: spawning processes and assembling the child process environment (vault
* injection + hardening).
* Docs: /docs/tools § "Background session caps".
*/
import { ManagedSession } from "./session.js";
import { BackgroundRegistry } from "../background/index.js";
/** Concurrent managed-session cap: evicts once exceeded (exited sessions first, otherwise LRU — killing a background process has bounded cost). */
const MAX_SESSIONS = 64;
/**
* Hardening overrides applied to the child process environment: suppresses editor/credential
* prompts/pagers/color etc. that could interact, avoiding a command hanging while waiting for
* input. `GIT_EDITOR=true` prevents `git commit`/`rebase -i` from popping an editor;
* `GIT_TERMINAL_PROMPT=0` prevents git from interactively asking for credentials; in pipe mode,
* git and similar tools already auto-disable the pager, so the `PAGER` entries are just an extra
* safeguard.
*/
const HARDENED_ENV: NodeJS.ProcessEnv = {
GIT_EDITOR: "true",
GIT_TERMINAL_PROMPT: "0",
TERM: "dumb",
NO_COLOR: "1",
PAGER: "cat",
GIT_PAGER: "cat",
};
export class CommandSessionManager {
private readonly registry = new BackgroundRegistry<ManagedSession>({
idPrefix: "proc",
maxTasks: MAX_SESSIONS,
});
/** Agent vault environment variables: injected into the child process on every spawn (values never enter the model context, only the environment). */
private readonly vault: Record<string, string>;
constructor(opts?: { vault?: Record<string, string> }) {
this.vault = opts?.vault ?? {};
}
/** Starts a command, returning an **unregistered** session (no process_id yet). */
spawn(opts: { cmd: string; cwd: string }): ManagedSession {
if (this.registry.isDisposed) {
throw new Error("command session manager disposed");
}
return new ManagedSession({
cmd: opts.cmd,
cwd: opts.cwd,
// Spread order is priority: vault overrides host variables of the same name, but must
// come before HARDENED_ENV — the hardening entries (GIT_EDITOR/PAGER etc. that prevent
// interactive hangs) must never be overridable by vault.
env: { ...process.env, ...this.vault, ...HARDENED_ENV },
});
}
/** Registers a still-running session as a background process, allocating and returning a unique `process_id`. */
register(session: ManagedSession): string {
this.registry.makeRoom(true);
return this.registry.register(session);
}
/** Looks up a session by process_id and refreshes its access time; returns undefined if it doesn't exist. */
get(processId: string): ManagedSession | undefined {
return this.registry.get(processId);
}
/** Removes from the registry and cleans up the process group (called after the session exits). */
remove(processId: string): void {
this.registry.remove(processId);
}
/** Disposes: removes the fallback registration and kills all sessions (the process 'exit' fallback is hooked up by the registry itself). Idempotent. */
dispose(): void {
this.registry.dispose();
}
}
@@ -0,0 +1,228 @@
/**
* ManagedSession — runtime state and collection logic for a single command session.
*
* Spawns the process with `bash -lc <cmd>`, with stdout/stderr going through plain pipes (no
* native dependency, clean output; an interactive program that detects no TTY falls back to
* non-interactive mode, which parses more cleanly for the Agent anyway). `detached` makes the
* child process the process-group leader, so both Ctrl-C and killing the whole group rely on
* **process-group signals** (sending a signal to `-pid` also reaches background child processes).
*
* Key semantics:
* - **Termination is determined by the foreground process exiting (the exit event, waitpid
* semantics), not by waiting for stream EOF**: background child processes that inherit the
* pipe don't hold things up;
* - `collect(yieldMs)` **streams** output deltas within the budget: data is yielded as soon as it
* arrives, without waiting for the window to end; if the command exits mid-window, the trailing
* output is yielded along with it (with a capped drain window); if it's still running once the
* window expires, whatever output exists is yielded and collection ends, with the process
* switching to background; if `signal` aborts, whatever output exists is yielded and collection
* ends immediately;
* - Unread output has a cap (memory safety); when exceeded, the oldest part is dropped and
* counted, with a marker shown on read;
* - `kill()` sends SIGTERM to the process group, then SIGKILL after a grace period, reaping any
* leftover background child processes; idempotent.
*/
import { spawn, type ChildProcess } from "node:child_process";
import type { ToolResult } from "../types.js";
import { CappedTextBuffer, WakeSignal } from "../background/index.js";
/** Process-group semantics are available on POSIX; Windows falls back to signaling the child process directly. */
const SUPPORTS_PROCESS_GROUP = process.platform !== "win32";
/** Extra wait cap (ms) after the command exits to collect trailing output: enough to drain the last flush, without hanging. */
const POST_EXIT_DRAIN_MS = 50;
/** Capacity cap (characters) for a single session's unread output: prevents a chatty background process from blowing up memory. */
const OUTPUT_BUFFER_CAP = 1024 * 1024; // 1 MiB
/**
* Grace period (ms) before escalating from SIGTERM to SIGKILL: gives a process that needs to
* clean up (flush data, remove temp files) some time. The timer is unref'd, so it won't hold up
* the host process from exiting; the process-exit path sends SIGKILL directly as a fallback.
*/
const SIGKILL_GRACE_MS = 1_000;
/** Foreground process exit info. At most one of `code`/`signal` is set (consistent with Node child's exit event). */
export interface ProcessExit {
code: number | null;
signal: NodeJS.Signals | null;
}
/** Arguments required to start a command. */
export interface SpawnOptions {
/** Command string handed to `bash -lc`. */
cmd: string;
/** Working directory (absolute path). */
cwd: string;
/** Child process environment variables (the caller has already injected hardening entries like PAGER/TERM). */
env: NodeJS.ProcessEnv;
}
export class ManagedSession {
/** Timestamp of the last access (used for LRU / idle reaping). */
lastUsed: number = Date.now();
private readonly child: ChildProcess;
private readonly buffer = new CappedTextBuffer(OUTPUT_BUFFER_CAP, "earlier output");
private exited = false;
private exitInfo: ProcessExit | null = null;
private spawnError: Error | null = null;
private killed = false;
private killTimer: ReturnType<typeof setTimeout> | null = null;
// Single wake point: data arrival / process exit / spawn error all wake a waiting collect through it.
private readonly wakeSignal = new WakeSignal();
constructor(opts: SpawnOptions) {
this.child = spawn("bash", ["-lc", opts.cmd], {
cwd: opts.cwd,
env: opts.env,
detached: SUPPORTS_PROCESS_GROUP, // Become the process-group leader, so the whole group can be signaled
stdio: ["pipe", "pipe", "pipe"],
});
this.child.stdout?.setEncoding("utf8");
this.child.stderr?.setEncoding("utf8");
// stdin may already be closed by the command before input_command writes to it;
// EPIPE/ERR_STREAM_DESTROYED are an expected race and must not bubble up to the host process
// as an unhandled error.
this.child.stdin?.on("error", () => {});
this.child.stdout?.on("data", (c: string) => this.handleData(c));
this.child.stderr?.on("data", (c: string) => this.handleData(c));
// exit follows waitpid semantics: it fires as soon as bash exits, without waiting for
// stdout/stderr pipe EOF — background child processes that inherit and hold the pipe open
// won't hold up termination.
this.child.on("exit", (code, signal) => this.handleExit({ code, signal }));
this.child.on("error", (err) => this.handleError(err));
}
/** Signals the process group; ignores the case where the process/group has already exited (ESRCH). */
private signalGroup(sig: NodeJS.Signals): void {
try {
if (SUPPORTS_PROCESS_GROUP && typeof this.child.pid === "number" && this.child.pid > 0) {
process.kill(-this.child.pid, sig); // Negative pid = the whole process group
} else {
this.child.kill(sig);
}
} catch {
// ESRCH etc., ignored.
}
}
private handleData(chunk: string): void {
this.buffer.append(chunk);
this.wakeSignal.notify();
}
private handleExit(exit: ProcessExit): void {
if (this.exited) return;
this.exited = true;
this.exitInfo = exit;
this.wakeSignal.notify();
}
private handleError(err: Error): void {
if (this.exited) return;
this.spawnError = err;
this.exited = true; // A spawn failure is also treated as a terminal state
this.wakeSignal.notify();
}
/** Whether the command is still running (hasn't exited, spawn hasn't failed). */
get running(): boolean {
return !this.exited;
}
get exit(): ProcessExit | null {
return this.exitInfo;
}
get error(): Error | null {
return this.spawnError;
}
/**
* Streams output deltas within `yieldMs` (data is yielded as soon as it arrives). Once done,
* the terminal state is determined via `running`/`exit`/`error`:
* - Exits mid-window -> the trailing output is yielded along with it (extra ≤POST_EXIT_DRAIN_MS drain);
* - Still running once the window expires -> whatever output exists is yielded and collection ends, with the process switching to background;
* - `signal` aborts -> whatever output exists is yielded and collection ends immediately (the process isn't killed; the caller decides whether to keep it).
*/
async *collect(yieldMs: number, signal?: AbortSignal): AsyncGenerator<string> {
const start = Date.now();
const onAbort = (): void => this.wakeSignal.notify();
signal?.addEventListener("abort", onAbort, { once: true });
try {
// Phase one: running, data is yielded as soon as it arrives, until exit / abort / yield expires.
while (!this.exited) {
const chunk = this.buffer.drain();
if (chunk) yield chunk;
if (signal?.aborted) return;
const remaining = yieldMs - (Date.now() - start);
if (remaining <= 0) {
const tail = this.buffer.drain();
if (tail) yield tail;
return; // Still running -> yield
}
// Re-check the predicate before sleeping: data that arrives while `yield` is suspended
// wakes at a point before this wait begins, and would otherwise be missed.
if (!this.buffer.isEmpty) continue;
await this.wakeSignal.wait(remaining);
}
// Phase two: already exited (or spawn failed) -> drain the trailing output, with a cap.
const head = this.buffer.drain();
if (head) yield head;
const drainStart = Date.now();
for (;;) {
if (!this.buffer.isEmpty) {
yield this.buffer.drain();
continue;
}
const left = POST_EXIT_DRAIN_MS - (Date.now() - drainStart);
if (left <= 0) break;
await this.wakeSignal.wait(left);
if (this.buffer.isEmpty) break; // Woke with no new data (or timed out) -> draining is done
}
const tail = this.buffer.drain();
if (tail) yield tail;
} finally {
signal?.removeEventListener("abort", onAbort);
}
}
write(chars: string): void {
this.lastUsed = Date.now();
try {
if (!this.child.stdin || this.child.stdin.destroyed) return;
this.child.stdin.write(chars, () => {});
} catch {
// stdin may already be closed, ignored.
}
}
interrupt(): void {
this.lastUsed = Date.now();
this.signalGroup("SIGINT");
}
/** Closes out: sends SIGTERM to the process group, then SIGKILL after a grace period (reaping leftover background child processes); idempotent. */
kill(): void {
if (this.killed) return;
this.killed = true;
this.signalGroup("SIGTERM");
// Unconditionally escalates to SIGKILL: the foreground has exited but background child
// processes may still be around; killpg on an already-vanished group is ESRCH (harmless).
this.killTimer = setTimeout(() => this.signalGroup("SIGKILL"), SIGKILL_GRACE_MS);
this.killTimer.unref?.();
}
/** Synchronous hard kill (process 'exit' fallback: the event loop has already stopped at this point, so timers aren't available). */
killHard(): void {
this.killed = true;
if (this.killTimer) {
clearTimeout(this.killTimer);
this.killTimer = null;
}
this.signalGroup("SIGKILL");
}
}
/** Converts exit info into a tool result (the terminal marker is appended via `note`, outside the truncation, so it isn't lost with long output). */
export function resultForExit(exit: ProcessExit | null): ToolResult {
if (!exit) return { stopReason: "completed" };
if (exit.signal) return { stopReason: "failed", note: `[terminated by signal ${exit.signal}]` };
if (exit.code !== 0)
return { stopReason: "failed", note: `[exit code: ${exit.code ?? "unknown"}]` };
return { stopReason: "completed" };
}
@@ -0,0 +1,115 @@
/**
* describe_image —— image-proxy-read tool, the text-only-model variant of read_image
* (`forModel: "text-only"`, see default-config.ts): the image itself is never fed back into
* the session model (some providers flatly 400 on a tool_result carrying an image); instead it
* is sent, together with the caller-supplied `prompt`, in a single one-off request to the
* Project-configured vision model (`vision_model`), and the vision model's text answer is
* returned as the tool's output.
*
* The tool definition (description/parameters, including `prompt`) comes entirely from the
* config entry; this implementation does no runtime rewriting — which tool is used for which
* model class is declared by the entry's `forModel` annotation, so the config file is what you get.
*
* Behavioral contract (shared with read_image): `source` supports http(s) URLs and local paths;
* validation/size limits are reused from `loadImage`; on failure, outputs explanatory text and
* finishes with `failed`, never throws; on interruption, only reports `aborted`. Messages from
* the internal one-off request never enter the parent session stream (no origin, not leaked out).
* Docs: /docs/tools § "Image tools".
*/
import { imageUrlMessage, partialToolCallOutput, userText } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { LLMOutcome, ToolDefinitionConfig, VisionDescriberService } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import { formatSize, loadImage } from "./read-image.js";
/** Tool name constant (used only inside this tool module, not exposed to Environment). */
export const DESCRIBE_IMAGE_NAME = "describe_image";
/** Default question used when the caller doesn't supply a prompt. */
const DEFAULT_PROMPT =
"Describe this image in detail, including any visible text, numbers, UI elements and layout.";
/** Constructs the describe_image tool: definition (description/parameters) is taken as-is from the config entry. */
export function createDescribeImageTool(
definition: ToolDefinitionConfig,
describer: VisionDescriberService,
): BuiltinTool {
return {
name: DESCRIBE_IMAGE_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
const source = args["source"];
if (typeof source !== "string" || source.length === 0) {
yield delta('Missing required argument "source" for describe_image.');
return { stopReason: "failed" };
}
if (describer.modelId === null || describer.createLLM === undefined) {
yield delta(
"No vision model is configured for this project. The current model does not accept " +
"images; ask the user to pick a vision model in the model settings (vision_model) " +
"to enable image reading.",
);
return { stopReason: "failed" };
}
const res = await loadImage(source, ctx.workspaceDir, signal);
if (!res.ok) {
if (res.reason === "aborted") return { stopReason: "aborted" };
yield delta(res.message);
return { stopReason: "failed" };
}
const prompt =
typeof args["prompt"] === "string" && args["prompt"].trim().length > 0
? args["prompt"]
: DEFAULT_PROMPT;
const dataUrl = `data:${res.mime};base64,${res.bytes.toString("base64")}`;
// One-off vision model request: prompt + image are merged into a single user message;
// its text deltas (partial_text delta) are forwarded in real time as this tool's own
// output delta — the description streams out piece by piece, not buffered as a whole.
// Partial concatenation == the full message (see generative-model.ts), so the full text
// is not forwarded again; other messages like thinking/token_usage are ignored, never
// leaked into the parent session.
const llm = describer.createLLM();
const gen = llm.streamGenerate({
newMessages: [userText(prompt), imageUrlMessage(dataUrl)],
...(signal ? { signal } : {}),
});
yield delta(
`${res.mime}, ${formatSize(res.bytes.length)} — described by ${describer.modelId}:\n`,
);
let streamedAny = false;
let outcome: LLMOutcome | undefined;
for (;;) {
const step = await gen.next();
if (step.done) {
outcome = step.value;
break;
}
const p = step.value.payload as { type?: string; event_type?: string; text?: string };
if (p.type === "partial_text" && p.event_type === "delta" && p.text) {
streamedAny = true;
yield delta(p.text);
}
}
if (signal?.aborted) return { stopReason: "aborted" };
if (!outcome || outcome.status !== "completed") {
const detail =
outcome && "message" in outcome && outcome.message ? `: ${outcome.message}` : "";
yield delta(
`${streamedAny ? "\n" : ""}Vision model (${describer.modelId}) request ${outcome?.status ?? "failed"}${detail}`,
);
return { stopReason: "failed" };
}
if (!streamedAny) yield delta("[vision model returned no text]");
},
};
}
@@ -0,0 +1,121 @@
/**
* exec_command —— local shell executor, a built-in tool implementation (BuiltinTool).
*
* Spawns a process inside the Workspace via `bash -lc <cmd>` and streams content deltas as
* stdout/stderr chunks arrive. Waits up to `yield_time_ms`: if the command finishes in time,
* returns the full output and exit status; if it's still running when the deadline hits,
* returns the output collected so far plus a `process_id` — the process moves to background,
* managed by `CommandSessionManager`, and is interacted with afterward via `input_command`.
* Completion is decided by **the foreground
* process exiting**, not by waiting for EOF on the output stream — background children (e.g.
* `node server.js &`) that inherit the pipes won't hold up the tool.
*
* Division of responsibility with Environment (see environment.ts): this tool only produces
* content deltas; exit code/signal/spawn errors and `process_id` are reported via the return
* value's `note` (appended outside the maxOutputLength truncation, so it survives even when
* long output gets truncated). Whether it ends normally or abnormally, it always finishes via
* the return value, **never throws**; on interruption it only reports `aborted` — the
* interruption note itself is appended by Environment.
* Docs: /docs/tools § "Command sessions".
*/
import path from "node:path";
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import { DEFAULT_EXEC_YIELD_MS, resultForExit } from "./command/index.js";
import { clampYield } from "./background/index.js";
/** Tool name constant (used only inside this tool module, not exposed to Environment). */
export const EXEC_COMMAND_NAME = "exec_command";
/**
* exec_command built-in tool: parses arguments, resolves workdir, and delegates to
* `CommandSessionManager` to spawn the process and collect output.
* `definition` is overridden by Environment at construction time with the matching entry
* from ToolConfig (description/parameters/permission/limits).
* `services.commandSessions` is injected by Environment (shares the same registry with
* input_command).
*/
export function createExecCommandTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const manager = services?.commandSessions;
return {
name: EXEC_COMMAND_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
if (!manager) {
yield delta("[exec_command unavailable: no command session manager configured]");
return { stopReason: "failed" };
}
const cmd = args["cmd"];
if (typeof cmd !== "string" || cmd.length === 0) {
yield delta('Missing required argument "cmd" for exec_command.');
return { stopReason: "failed" };
}
// workdir defaults to workspaceDir; relative paths are resolved against workspaceDir.
const rawWorkdir = args["workdir"];
const workdir =
typeof rawWorkdir === "string" && rawWorkdir.length > 0
? path.resolve(ctx.workspaceDir, rawWorkdir)
: ctx.workspaceDir;
const yieldMs = clampYield(
args["yield_time_ms"],
DEFAULT_EXEC_YIELD_MS,
definition.timeoutMs,
);
// Caller already aborted: finish immediately with aborted.
if (signal?.aborted) return { stopReason: "aborted" };
let session;
try {
session = manager.spawn({ cmd, cwd: workdir });
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
yield delta(`[spawn error: ${message}]`);
return { stopReason: "failed" };
}
// On interruption, kill the whole process group (background children included) to
// avoid orphans; once the process moves to background this listener is removed in finally.
const onAbort = (): void => session.kill();
let registered = false;
signal?.addEventListener("abort", onAbort, { once: true });
try {
for await (const chunk of session.collect(yieldMs, signal)) yield delta(chunk);
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
// Still running at the deadline: register as a background process, returning
// process_id so input_command can continue accessing it.
const id = manager.register(session);
registered = true;
return {
stopReason: "completed",
note: `[process running with process_id ${id}; use input_command to send input or poll for output]`,
};
}
// Already exited: report exit status; process group cleanup (reaping any leftover
// background children) is handled uniformly in finally.
if (session.error) {
return { stopReason: "failed", note: `[spawn error: ${session.error.message}]` };
}
return resultForExit(session.exit);
} finally {
signal?.removeEventListener("abort", onAbort);
if (!registered) session.kill();
}
},
};
}
@@ -0,0 +1,109 @@
/**
* input_command — accesses a long-running command session started by `exec_command` (BuiltinTool).
*
* Finds the session by `process_id`: if `chars` is non-empty, writes it to stdin first (when it is exactly `\u0003`, special-cased as sending SIGINT to the
* process group, i.e. Ctrl-C — it must be sent alone; mixing it with other content errors out,
* since a pipe has no terminal line discipline and a mixed-in ETX byte would just be written
* into stdin silently with no effect), and if empty, nothing is written and it only polls.
* It then collects new output within `yield_time_ms` or waits for exit. If the command is still
* running, returns the same `process_id`; once exited, returns the trailing output and exit
* status and cleans up the session.
*
* Shares the same `CommandSessionManager` injected by Environment with exec_command. An
* interruption only cancels this poll — **it does not kill the background process** (the process
* was started independently earlier; interrupting one poll shouldn't kill it as a side effect).
* Docs: /docs/tools § "Command sessions".
*/
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import {
DEFAULT_EMPTY_POLL_YIELD_MS,
DEFAULT_WRITE_YIELD_MS,
resultForExit,
} from "./command/index.js";
import { clampYield } from "./background/index.js";
/** Tool name constant. */
export const INPUT_COMMAND_NAME = "input_command";
/** Ctrl-C: the ETX control character (U+0003). Received alone, it sends SIGINT to the process group instead of writing the byte into stdin. */
const INTERRUPT = String.fromCharCode(3); // U+0003 (ETX)
export function createInputCommandTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const manager = services?.commandSessions;
return {
name: INPUT_COMMAND_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
if (!manager) {
yield delta("[input_command unavailable: no command session manager configured]");
return { stopReason: "failed" };
}
const processId = args["process_id"];
if (typeof processId !== "string" || processId.length === 0) {
yield delta('Missing required argument "process_id" for input_command.');
return { stopReason: "failed" };
}
const session = manager.get(processId);
if (!session) {
yield delta(
`[input_command error: unknown process_id ${processId} (the session may have exited and been cleared)]`,
);
return { stopReason: "failed" };
}
const chars = typeof args["chars"] === "string" ? (args["chars"] as string) : "";
const empty = chars.length === 0;
const yieldMs = clampYield(
args["yield_time_ms"],
empty ? DEFAULT_EMPTY_POLL_YIELD_MS : DEFAULT_WRITE_YIELD_MS,
definition.timeoutMs,
);
if (signal?.aborted) return { stopReason: "aborted" };
// Write / interrupt (empty chars just polls). U+0003 mixed with other content errors out
// rather than being written silently (same as codex): a pipe has no terminal line
// discipline, so an ETX byte in stdin produces no interruption — the model would just
// see the command still running.
if (!empty) {
if (chars === INTERRUPT) session.interrupt();
else if (chars.includes(INTERRUPT)) {
yield delta(
'[input_command error: chars mixes U+0003 (Ctrl-C) with other content; send "\\u0003" alone to interrupt]',
);
return { stopReason: "failed" };
} else session.write(chars);
}
for await (const chunk of session.collect(yieldMs, signal)) yield delta(chunk);
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
return {
stopReason: "completed",
note: `[process still running with process_id ${processId}]`,
};
}
// Already exited: clean up the registry and report the exit status.
manager.remove(processId);
if (session.error) {
return { stopReason: "failed", note: `[spawn error: ${session.error.message}]` };
}
return resultForExit(session.exit);
},
};
}
@@ -0,0 +1,125 @@
/**
* input_subagent —— accesses a subagent session that `run_subagent` moved to the background
* (BuiltinTool).
*
* Finds the session by `subagent_id`: when `prompt` is empty, nothing is written — it just
* polls (collecting subagent messages and text deltas buffered during the background period,
* or waiting for the run to end); when non-empty and the subagent is idle, it's fed in as a new
* user message to continue on the same child Session (long-running subagent, multi-turn
* conversation); when non-empty but the subagent is still running, it errors, suggesting to
* poll first. Within the window, queued approval requests from the child session are also
* passed through (see subagent/session.ts).
*
* Difference from `input_command`: once a round of work finishes, the session is **not
* removed** (kept to receive a follow-up prompt) — it's only released when the parent Session
* ends, or evicted as an idle session once concurrency is full. Interruption only aborts this
* particular access, **it never kills the child session** (the subagent was launched
* independently earlier; the user interrupting one poll shouldn't kill it along the way).
* Docs: /docs/tools § "Subagents".
*/
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import {
DEFAULT_SUBAGENT_POLL_YIELD_MS,
DEFAULT_SUBAGENT_YIELD_MS,
resultForSubagentExit,
} from "./subagent/index.js";
import { approvalHint } from "./run-subagent.js";
import { collectWindow } from "./subagent/collect.js";
import { clampYield } from "./background/index.js";
/** Tool name constant. */
export const INPUT_SUBAGENT_NAME = "input_subagent";
export function createInputSubagentTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const manager = services?.subagentSessions;
return {
name: INPUT_SUBAGENT_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal, approve } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
if (!manager) {
yield delta("[input_subagent unavailable: no subagent session manager configured]");
return { stopReason: "failed" };
}
const subagentId = args["subagent_id"];
if (typeof subagentId !== "string" || subagentId.length === 0) {
yield delta('Missing required argument "subagent_id" for input_subagent.');
return { stopReason: "failed" };
}
const session = manager.get(subagentId);
if (!session) {
yield delta(
`[input_subagent error: unknown subagent_id ${subagentId} ` +
`(the session may have finished and been cleared)]`,
);
return { stopReason: "failed" };
}
const prompt = typeof args["prompt"] === "string" ? (args["prompt"] as string) : "";
const empty = prompt.trim().length === 0;
const yieldMs = clampYield(
args["yield_time_ms"],
empty ? DEFAULT_SUBAGENT_POLL_YIELD_MS : DEFAULT_SUBAGENT_YIELD_MS,
definition.timeoutMs,
);
if (signal?.aborted) return { stopReason: "aborted" };
// Continue with a follow-up prompt (empty prompt just polls). New input is not accepted
// while running: poll first to collect progress.
if (!empty) {
if (session.running) {
yield delta(
`[input_subagent error: subagent ${subagentId} is still running; ` +
`poll with an empty prompt to collect progress first]`,
);
return { stopReason: "failed" };
}
// startRun expresses edge cases like already-disposed via throw, collapsed here into failed (the tool never throws outward).
try {
session.startRun(prompt);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
yield delta(`[input_subagent error: ${message}]`);
return { stopReason: "failed" };
}
}
yield* collectWindow(session, {
yieldMs,
toolCallId,
...(signal ? { signal } : {}),
...(approve ? { approve } : {}),
});
// Interruption only aborts this access, it doesn't kill the child session.
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
return {
stopReason: "completed",
note: `[subagent still running with subagent_id ${subagentId}]` + approvalHint(session),
};
}
// This round of work has ended: report the end state; the session is kept (can be resumed), not removed from the registry.
const result = resultForSubagentExit(session.exit);
const idleHint = `[subagent idle with subagent_id ${subagentId}; send a follow-up prompt to continue]`;
return {
...result,
note: result.note !== undefined ? `${result.note} ${idleHint}` : idleHint,
};
},
};
}
@@ -0,0 +1,248 @@
/**
* read_image — image-reading tool, a builtin tool implementation (BuiltinTool).
*
* Reads an image and feeds it back to the model as **image content**: if `source` is an http(s)
* URL, downloads it with the global fetch (respecting the abort signal); otherwise reads it as a
* local file path (relative paths are resolved against the Workspace). Only png/jpeg/gif/webp
* are allowed (determined in order by response header / magic number / extension); errors out
* above 5MB.
*
* Division of responsibility with Environment (see environment.ts): on success, yields a brief
* descriptive delta (e.g. `image/png, 123.4 kB`), while the image itself is carried via the
* return value `ToolResult.images` (a data URL) for Environment to attach when closing out (a
* single streaming delta carries it all at once before stop, plus the final complete
* `tool_call_output`); on failure, yields explanatory text and closes with `failed`, **never
* throwing**; if interrupted, only reports `aborted` — the interruption note is appended by
* Environment.
*
* This tool is only used by sessions with a model that supports images (config entry
* `forModel: "vision"`); text-only models use describe_image instead (the image is handed to a
* configured vision model to describe, returning text — see describe-image.ts), and the image
* loading/validation logic is shared via `loadImage`.
* Docs: /docs/tools § "Image tools".
*/
import path from "node:path";
import { readFile, stat } from "node:fs/promises";
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
/** Tool name constant (used only within this tool module, never exposed to Environment). */
export const READ_IMAGE_NAME = "read_image";
/**
* Image size upper bound (bytes): errors out above this. Taken as the common denominator of
* per-provider single-image hard limits (Claude API is around 5MB, some compatible endpoints are
* lower) — since local validation passing but the next request getting a blanket 400 from the
* provider is a non-retryable path, the limit must not exceed the strictest downstream; this also
* avoids oversized images blowing up the context and Trace.
*/
export const MAX_IMAGE_BYTES = 5 * 1024 * 1024;
/** Supported image mime types (the four generally accepted across providers). */
const SUPPORTED_MIMES = new Set(["image/png", "image/jpeg", "image/gif", "image/webp"]);
/** Extension -> mime (fallback when magic-number sniffing fails). */
const EXT_TO_MIME: Record<string, string> = {
".png": "image/png",
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".gif": "image/gif",
".webp": "image/webp",
};
/** Sniffs the mime type from the file header's magic number; returns null if unrecognized. */
function sniffMime(buf: Buffer): string | null {
if (buf.length >= 8 && buf.readUInt32BE(0) === 0x89504e47) return "image/png";
if (buf.length >= 3 && buf[0] === 0xff && buf[1] === 0xd8 && buf[2] === 0xff) {
return "image/jpeg";
}
if (buf.length >= 6) {
const head = buf.subarray(0, 6).toString("latin1");
if (head === "GIF87a" || head === "GIF89a") return "image/gif";
}
if (
buf.length >= 12 &&
buf.subarray(0, 4).toString("latin1") === "RIFF" &&
buf.subarray(8, 12).toString("latin1") === "WEBP"
) {
return "image/webp";
}
return null;
}
/** Infers the mime type from a path / URL pathname's extension; returns null if it can't be inferred. */
function mimeFromExt(p: string): string | null {
return EXT_TO_MIME[path.extname(p).toLowerCase()] ?? null;
}
/** Byte count -> human-readable size (B / kB / MB, one decimal place). */
export function formatSize(bytes: number): string {
if (bytes < 1024) return `${bytes} B`;
const kb = bytes / 1024;
if (kb < 1024) return `${kb.toFixed(1)} kB`;
return `${(kb / 1024).toFixed(1)} MB`;
}
const OVERSIZE_MESSAGE = (size: number): string =>
`Image too large: ${formatSize(size)} exceeds the ${formatSize(MAX_IMAGE_BYTES)} limit.`;
const UNSUPPORTED_MESSAGE = (detected: string | null): string =>
`Unsupported image type${detected ? ` "${detected}"` : ""}: only png, jpeg, gif and webp are supported.`;
/** Result of `loadImage`: success (bytes + mime) / interrupted / failed (explanatory message). */
export type LoadImageResult =
| { ok: true; bytes: Buffer; mime: string }
| { ok: false; reason: "aborted" }
| { ok: false; reason: "failed"; message: string };
/**
* Reads and validates an image (shared by read_image and describe_image):
* an http(s) URL is downloaded with the global fetch, otherwise read as a local path (resolved
* against Workspace); validates the size upper bound and mime type (determined in order by
* response header / magic number / extension). Never throws.
*/
export async function loadImage(
source: string,
workspaceDir: string,
signal?: AbortSignal,
): Promise<LoadImageResult> {
if (signal?.aborted) return { ok: false, reason: "aborted" };
let bytes: Buffer;
let mime: string | null;
if (/^https?:\/\//i.test(source)) {
// URL branch: downloads via the global fetch (abort signal passed through to the request);
// mime is preferentially taken from the response header, falling back to magic number / URL
// extension.
let res: Response;
try {
res = await fetch(source, signal ? { signal } : {});
} catch (err) {
if (signal?.aborted) return { ok: false, reason: "aborted" };
const message = err instanceof Error ? err.message : String(err);
return {
ok: false,
reason: "failed",
message: `Failed to download image "${source}": ${message}`,
};
}
if (!res.ok) {
return {
ok: false,
reason: "failed",
message: `Failed to download image "${source}": HTTP ${res.status}`,
};
}
// When content-length is trustworthy, reject an oversized response early to avoid reading it
// into memory for nothing.
const declared = Number(res.headers.get("content-length") ?? "");
if (Number.isFinite(declared) && declared > MAX_IMAGE_BYTES) {
return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(declared) };
}
try {
bytes = Buffer.from(await res.arrayBuffer());
} catch (err) {
if (signal?.aborted) return { ok: false, reason: "aborted" };
const message = err instanceof Error ? err.message : String(err);
return {
ok: false,
reason: "failed",
message: `Failed to download image "${source}": ${message}`,
};
}
const headerMime = (res.headers.get("content-type") ?? "").split(";")[0]!.trim().toLowerCase();
let urlExtMime: string | null = null;
try {
urlExtMime = mimeFromExt(new URL(source).pathname);
} catch {
urlExtMime = null; // A URL parse failure only affects the extension fallback
}
mime = SUPPORTED_MIMES.has(headerMime) ? headerMime : (sniffMime(bytes) ?? urlExtMime);
if (mime === null && headerMime) mime = headerMime; // Include the real response type in the error
} else {
// Local-path branch: relative paths are resolved against Workspace; stat first to check the
// size before reading, to avoid reading an oversized file into memory in one go.
const filePath = path.resolve(workspaceDir, source);
try {
const st = await stat(filePath);
// Explicitly reject non-file paths such as directories: readFile's EISDIR error isn't
// model-friendly.
if (!st.isFile()) {
return {
ok: false,
reason: "failed",
message: `Failed to read image "${source}": path is not a file.`,
};
}
if (st.size > MAX_IMAGE_BYTES) {
return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(st.size) };
}
bytes = await readFile(filePath);
} catch (err) {
if (signal?.aborted) return { ok: false, reason: "aborted" };
const message = err instanceof Error ? err.message : String(err);
return {
ok: false,
reason: "failed",
message: `Failed to read image "${source}": ${message}`,
};
}
mime = sniffMime(bytes) ?? mimeFromExt(filePath);
}
if (signal?.aborted) return { ok: false, reason: "aborted" };
// Empty file/response: magic-number sniffing fails to identify it, but the extension fallback
// may still let it through — an empty base64 sent to the provider is guaranteed to error, so
// reject it here.
if (bytes.length === 0) {
return { ok: false, reason: "failed", message: `Image "${source}" is empty.` };
}
if (bytes.length > MAX_IMAGE_BYTES) {
return { ok: false, reason: "failed", message: OVERSIZE_MESSAGE(bytes.length) };
}
if (mime === null || !SUPPORTED_MIMES.has(mime)) {
return { ok: false, reason: "failed", message: UNSUPPORTED_MESSAGE(mime) };
}
return { ok: true, bytes, mime };
}
/**
* read_image builtin tool: reads a local file or downloads a URL, validates its type and size,
* then outputs a data URL image. `definition` is overridden by Environment at construction time
* with the same-named entry from ToolConfig (description/arguments/permissions/limits).
*/
export function createReadImageTool(definition: ToolDefinitionConfig): BuiltinTool {
return {
name: READ_IMAGE_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal } = ctx;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
const source = args["source"];
if (typeof source !== "string" || source.length === 0) {
yield delta('Missing required argument "source" for read_image.');
return { stopReason: "failed" };
}
const res = await loadImage(source, ctx.workspaceDir, signal);
if (!res.ok) {
if (res.reason === "aborted") return { stopReason: "aborted" };
yield delta(res.message);
return { stopReason: "failed" };
}
// Success: yield a brief one-line description as a text delta (both in the streaming and
// complete message), while the image itself is carried via the return value for
// Environment to attach.
yield delta(`${res.mime}, ${formatSize(res.bytes.length)}`);
return { images: [`data:${res.mime};base64,${res.bytes.toString("base64")}`] };
},
};
}
@@ -0,0 +1,47 @@
/**
* Built-in tool registry —— maps tool names to BuiltinTool factories.
*
* Environment uses this table to assemble entries from ToolConfig into BuiltinTool instances:
* a tool is only assembled if its name is in the table (i.e. a supported built-in tool); the
* description/parameters/permission/maxOutputLength from config are injected into the tool's
* `definition` by each factory.
* When adding a new built-in tool, just register one factory entry here — no changes to
* Environment needed.
*
* Docs: packages/docs/content/tools.{zh,en}.md (site path /docs/tools) documents every
* built-in tool and the approval flow — keep the page in sync when this table changes.
*/
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool } from "./types.js";
import { EXEC_COMMAND_NAME, createExecCommandTool } from "./exec-command.js";
import { INPUT_COMMAND_NAME, createInputCommandTool } from "./input-command.js";
import { SUBAGENT_NAME, createSubagentTool } from "./run-subagent.js";
import { INPUT_SUBAGENT_NAME, createInputSubagentTool } from "./input-subagent.js";
import { READ_IMAGE_NAME, createReadImageTool } from "./read-image.js";
import { DESCRIBE_IMAGE_NAME, createDescribeImageTool } from "./describe-image.js";
/**
* A factory that constructs a BuiltinTool instance from a tool config entry; optionally
* receives runtime services injected by Environment.
* Most tools ignore `services`; only a few (e.g. `run_subagent`) use it.
*/
export type BuiltinToolFactory = (
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
) => BuiltinTool;
/** Tool name -> factory. */
export const BUILTIN_TOOL_FACTORIES: Record<string, BuiltinToolFactory> = {
[EXEC_COMMAND_NAME]: createExecCommandTool,
[INPUT_COMMAND_NAME]: createInputCommandTool,
[SUBAGENT_NAME]: createSubagentTool,
[INPUT_SUBAGENT_NAME]: createInputSubagentTool,
[READ_IMAGE_NAME]: createReadImageTool,
// describe_image: the text-only-model variant of read_image (hands the image to the
// configured vision model for description, returns text).
// Which tool is used for which model class is declared by the config entry's forModel
// annotation; before assembly, selectBuiltinToolsForModel has already filtered out entries
// that don't apply to the session's model.
[DESCRIBE_IMAGE_NAME]: (definition, services) =>
createDescribeImageTool(definition, services?.visionDescriber ?? { modelId: null }),
};
@@ -0,0 +1,154 @@
/**
* run_subagent — delegates a subtask to a child Agent, supporting a switch to long-running
* background execution.
*
* The tool itself doesn't depend on Agent/Session, only holding an injected `SubagentRunner`
* (breaking the circular dependency). The model may freely choose the child Agent (`agent_id`)
* and model (`model_id`) via arguments; if omitted, it falls back to reusing the current Agent
* and the Project's default Model respectively. The spawned child session is managed by
* `ManagedSubagentSession` (sharing the `SubagentSessionManager` injected by Environment with
* `input_subagent`).
*
* The two-phase semantics mirror `exec_command`: within the `yield_time_ms` window, child-session
* messages are forwarded live (tagged with origin, so the frontend can see the child Agent's tool
* calls and token usage) and the child Agent's text deltas are copied as this tool's output; if
* the child Agent finishes within the window, its terminal state is returned and the child
* session is released; if it's still running once the window expires, it's registered as a
* background session, returning `subagent_id` for subsequent access the same way as
* `input_command` (polling / appending a Prompt, see input-subagent.ts).
*
* Approval: `run_subagent` itself is a read-write tool (`rw`), so its invocation requires Human
* approval; the child session's tool approval requests are forwarded to the same Human within
* the window via the session's approval queue (tagged with origin), and queued for the next
* access while running in the background. An interruption within the startup window kills the
* child session per exec_command semantics; precheck errors such as exceeding the depth limit or
* a nonexistent agent are expressed by the runner as a throw, and collapsed to failed.
* Docs: /docs/tools § "Subagents".
*/
import { partialToolCallOutput } from "../../omnimessage/index.js";
import type { OmniMessage } from "../../omnimessage/index.js";
import type { EnvironmentServices, ToolDefinitionConfig } from "../../interfaces.js";
import type { BuiltinTool, ToolExecutionContext, ToolResult } from "./types.js";
import {
DEFAULT_SUBAGENT_YIELD_MS,
ManagedSubagentSession,
resultForSubagentExit,
} from "./subagent/index.js";
import { collectWindow } from "./subagent/collect.js";
import { clampYield } from "./background/index.js";
/** Tool name constant (used only within this tool module, never exposed to Environment). */
export const SUBAGENT_NAME = "run_subagent";
/** Pending-approval hint: lets the model know it should poll again to move the child Agent forward. */
export function approvalHint(session: ManagedSubagentSession): string {
const n = session.pendingApprovals;
return n > 0 ? ` [subagent is waiting for approval of ${n} tool call(s); poll to review]` : "";
}
/** Builds run_subagent's BuiltinTool from tool config + injected services. */
export function createSubagentTool(
definition: ToolDefinitionConfig,
services?: EnvironmentServices,
): BuiltinTool {
const runner = services?.subagentRunner;
const manager = services?.subagentSessions;
return {
name: SUBAGENT_NAME,
definition,
async *execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void> {
const { toolCallId, signal, approve } = ctx;
const fail = function* (msg: string): Generator<OmniMessage> {
yield partialToolCallOutput({ eventType: "delta", output: msg, toolCallId });
};
// Missing arguments / unconfigured services both collapse to an explanatory output rather
// than throwing (consistent with other tools).
if (!runner) {
yield* fail("[run_subagent unavailable: no subagent runner configured]");
return { stopReason: "failed" };
}
if (!manager || manager.isDisposed) {
yield* fail("[run_subagent unavailable: no subagent session manager available]");
return { stopReason: "failed" };
}
const prompt = typeof args.prompt === "string" ? args.prompt : "";
if (prompt.trim().length === 0) {
yield* fail("[run_subagent error: missing required string argument `prompt`]");
return { stopReason: "failed" };
}
const agentId = typeof args.agent_id === "string" ? args.agent_id : undefined;
const modelId = typeof args.model_id === "string" ? args.model_id : undefined;
const yieldMs = clampYield(
args.yield_time_ms,
DEFAULT_SUBAGENT_YIELD_MS,
definition.timeoutMs,
);
if (signal?.aborted) return { stopReason: "aborted" };
// Concurrency cap (running child Agents are never evicted): reject spawning if there's no
// room.
if (!manager.makeRoom()) {
yield* fail(
"[run_subagent error: too many background subagents; poll or finish existing ones first]",
);
return { stopReason: "failed" };
}
// Spawn the child Session (precheck errors such as exceeding the depth limit or a
// nonexistent agent are expressed as a throw).
let session: ManagedSubagentSession;
try {
const handle = await runner.spawn({
...(agentId !== undefined ? { agentId } : {}),
...(modelId !== undefined ? { modelId } : {}),
});
session = new ManagedSubagentSession(handle);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
yield* fail(`[run_subagent error: ${message}]`);
return { stopReason: "failed" };
}
// An interruption within the startup window kills the child session (consistent with
// exec_command); once switched to background, this listener is removed in `finally`.
const onAbort = (): void => session.kill();
let registered = false;
signal?.addEventListener("abort", onAbort, { once: true });
try {
session.startRun(prompt);
yield* collectWindow(session, {
yieldMs,
toolCallId,
...(signal ? { signal } : {}),
...(approve ? { approve } : {}),
});
if (signal?.aborted) return { stopReason: "aborted" };
if (session.running) {
// Still running once the window expires: register as a background session, returning
// subagent_id for input_subagent to continue accessing it.
const id = manager.register(session);
registered = true;
return {
stopReason: "completed",
note:
`[subagent running with subagent_id ${id}; use input_subagent to poll for progress ` +
`or send a follow-up prompt]` +
approvalHint(session),
};
}
// Finished within the window: report the terminal state; releasing the child session is
// handled uniformly in `finally` (never registered, so no subagent_id).
return resultForSubagentExit(session.exit);
} finally {
signal?.removeEventListener("abort", onAbort);
if (!registered) session.kill();
}
},
};
}
@@ -0,0 +1,52 @@
/**
* collectWindow —— yield-window collector shared by run_subagent / input_subagent.
*
* Within the `yieldMs` window, emits child-session output in real time: buffered child-session
* messages (already origin-tagged, passed through to the frontend) and subagent text deltas
* (fed back to the LLM as this parent tool's own output delta). The window also hooks up an
* approval outlet, forwarding the child session's queued approval requests one by one to the
* Human via `approve`. The window ends on "run finished / signal abort / deadline reached", and
* does a final drain right before ending (to catch the tail buffer at the moment the run
* finishes). Deciding the end state and finalizing are the caller's responsibility.
*/
import { partialToolCallOutput } from "../../../omnimessage/index.js";
import type { OmniMessage } from "../../../omnimessage/index.js";
import type { ApproveFn } from "../../../interfaces.js";
import type { ManagedSubagentSession } from "./session.js";
export async function* collectWindow(
session: ManagedSubagentSession,
opts: { yieldMs: number; toolCallId: string; signal?: AbortSignal; approve?: ApproveFn },
): AsyncGenerator<OmniMessage> {
const { yieldMs, toolCallId, signal, approve } = opts;
const delta = (output: string): OmniMessage =>
partialToolCallOutput({ eventType: "delta", output, toolCallId });
const detach = approve ? session.attachApprovalSink(approve) : null;
// abort only ends this window (whether to kill the child session is up to the caller);
// wakes up a pending waitWake so it returns immediately.
const onAbort = (): void => session.wakeup();
signal?.addEventListener("abort", onAbort, { once: true });
try {
const start = Date.now();
for (;;) {
for (const m of session.drainMessages()) yield m;
const text = session.drainText();
if (text) yield delta(text);
if (!session.running) break;
if (signal?.aborted) break;
const remaining = yieldMs - (Date.now() - start);
if (remaining <= 0) break;
// Re-check the predicate before sleeping: output arriving while `yield` is suspended
// would fire its wakeup before this wait even starts, and get missed otherwise.
if (session.hasPending) continue;
await session.waitWake(remaining);
}
// Final drain: there may still be a tail buffer right when the run finishes/yields.
for (const m of session.drainMessages()) yield m;
const tail = session.drainText();
if (tail) yield delta(tail);
} finally {
signal?.removeEventListener("abort", onAbort);
detach?.();
}
}
@@ -0,0 +1,7 @@
/**
* Barrel for the background subagent session module.
*/
export { SubagentSessionManager } from "./session-manager.js";
export { ManagedSubagentSession, resultForSubagentExit } from "./session.js";
export type { SubagentExit } from "./session.js";
export { DEFAULT_SUBAGENT_YIELD_MS, DEFAULT_SUBAGENT_POLL_YIELD_MS } from "./limits.js";
@@ -0,0 +1,11 @@
/**
* Default yield duration for background subagent sessions.
*
* See `../background/limits.ts` for the clamping logic: only a lower bound is set, and the
* upper bound is derived from the tool's own `timeoutMs`.
*/
/** Default wait duration (ms) for `run_subagent` launching a task and `input_subagent` appending a Prompt to continue. */
export const DEFAULT_SUBAGENT_YIELD_MS = 300_000;
/** Default wait duration (ms) for `input_subagent` empty polling. */
export const DEFAULT_SUBAGENT_POLL_YIELD_MS = 10_000;
@@ -0,0 +1,63 @@
/**
* SubagentSessionManager —— registry and lifecycle management for background subagent sessions.
*
* Constructed by Environment (one per Session), injected via services to be shared by the
* `run_subagent` and `input_subagent` tools. Registry duties are handled by the generic
* `BackgroundRegistry` (shared with command sessions, see `../background/registry.ts`).
* Difference from command sessions: when at capacity, **running sessions are never evicted**
* (discarding in-progress subagent work is unacceptable) — only completed, idle ones are
* evicted; if there's still no room, the tool rejects spawning a new one.
* Docs: /docs/tools § "Background session caps".
*/
import { BackgroundRegistry } from "../background/index.js";
import type { ManagedSubagentSession } from "./session.js";
/**
* Cap on concurrently managed background subagent sessions. This is a **spawn admission cap**,
* not a hard limit: there's an await between the `makeRoom` check (before spawn) and `register`
* (after the yield window ends), so parallel run_subagent calls can briefly push the registered
* count over the cap — an already-running child session is never discarded just to hold the line.
*/
const MAX_SESSIONS = 8;
export class SubagentSessionManager {
private readonly registry = new BackgroundRegistry<ManagedSubagentSession>({
idPrefix: "subagent",
maxTasks: MAX_SESSIONS,
});
/** Whether the manager has been disposed (the host Session has ended). */
get isDisposed(): boolean {
return this.registry.isDisposed;
}
/** Whether there's still room for a new background session (evicting a completed, idle one if needed; never evicts a running one). */
makeRoom(): boolean {
return this.registry.makeRoom(false);
}
/**
* Registers a still-running session as a background session, allocating and returning a
* unique `subagent_id`: `subagent-<last 8 hex of child Session id>` (falls back to random on
* collision), whose suffix aligns with the message origin/frontend nesting label
* (`agent-<last 3 chars>`) for correlation.
*/
register(session: ManagedSubagentSession): string {
// A full yield window has elapsed since the pre-spawn makeRoom check, so the registry may
// have been filled by parallel calls in the meantime: free up room once more (only evicting
// completed, idle ones); if still no room, register anyway, tolerating a brief overshoot
// (see MAX_SESSIONS).
this.registry.makeRoom(false);
return this.registry.register(session, session.sessionId.slice(-8));
}
/** Looks up a session by subagent_id and refreshes its access time; returns undefined if not found. */
get(subagentId: string): ManagedSubagentSession | undefined {
return this.registry.get(subagentId);
}
/** Disposes: removes the fallback registration and finalizes all sessions (the process 'exit' fallback is hooked by the registry itself). Idempotent. */
dispose(): void {
this.registry.dispose();
}
}
@@ -0,0 +1,304 @@
/**
* ManagedSubagentSession — a subagent session capable of running in the background.
*
* Holds a `SubagentHandle` and drives its `run` (pump): the same child Session may run across
* multiple rounds (the first round is initiated by `run_subagent`, later rounds append a Prompt
* via `input_subagent`). Structurally mirrors a command session (ManagedSession): the parent tool
* call collects output live within the yield window; output produced outside the window (while
* running in the background) goes into a buffer, delivered all at once on the next access.
*
* Three kinds of output, each with its own destination:
* - **Message buffer**: all of the child session's OmniMessage (already tagged with origin), for
* the parent tool call to forward to the frontend for rendering; capped in count, overflow
* drops the oldest (only affects frontend replay — the child Session's own Trace loses no
* data);
* - **Text buffer**: assistant text deltas from the direct child layer (origin one hop), fed back
* as the parent tool's own output to the LLM; capped in capacity (prevents memory bloat),
* overflow drops the oldest with a marker;
* - **Approval queue**: the child session's tool approval requests. While running in the
* background, the parent session may have no active tool call to forward approval through, so
* the request is queued and the child session blocks waiting; the parent tool call
* (run_subagent / input_subagent) attaches an approval sink (`attachApprovalSink`) within its
* window to consult Human one request at a time — if detached mid-consultation (window ends),
* the request stays queued, and a late-arriving decision still takes effect (settled guard,
* first to arrive wins).
*
* Cleanup: `kill()` aborts the current run via AbortSignal, denies all pending approvals, and
* releases child Session resources; idempotent. The child Session runs in-process, so there's no
* need for a separate synchronous hard-kill path (its command sessions are reaped by their own
* exit fallback); `killHard` is equivalent to `kill`.
*/
import type { OmniMessage } from "../../../omnimessage/index.js";
import type { ApprovalDecision, ToolCallPayload } from "../../../omnimessage/index.js";
import type { ApproveFn, SubagentHandle } from "../../../interfaces.js";
import type { ToolResult } from "../types.js";
import { CappedTextBuffer, WakeSignal } from "../background/index.js";
/** Message buffer count cap: overflow drops the oldest (only frontend replay is affected — the child Session has its own Trace). */
const MESSAGE_BUFFER_CAP = 4096;
/** Text buffer capacity cap (characters): prevents a chatty child Agent from blowing up memory. */
const OUTPUT_BUFFER_CAP = 1024 * 1024; // 1 MiB
/** Terminal state of one run. */
export interface SubagentExit {
status: "completed" | "failed";
note?: string;
}
/** A pending approval request: the settled guard makes the decision first-to-arrive-wins (late/duplicate decisions are ignored). */
interface PendingApproval {
toolCall: OmniMessage<ToolCallPayload>;
settled: boolean;
resolve: (decision: ApprovalDecision) => void;
}
export class ManagedSubagentSession {
/** Timestamp of the last access (used for the eviction policy). */
lastUsed: number = Date.now();
private readonly handle: SubagentHandle;
private readonly abortCtrl = new AbortController();
private messages: OmniMessage[] = [];
private readonly textBuffer = new CappedTextBuffer(OUTPUT_BUFFER_CAP, "earlier subagent output");
private isRunning = false;
private exitInfo: SubagentExit | null = null;
private killed = false;
private readonly approvals: PendingApproval[] = [];
private sink: { approve: ApproveFn; detached: Promise<void> } | null = null;
private sinkEpoch = 0;
private pumpingApprovals = false;
// Single wake point: new message / run finished / new approval request all wake a waiting waitWake through it.
private readonly wakeSignal = new WakeSignal();
constructor(handle: SubagentHandle) {
this.handle = handle;
}
/** Child Session id (one hop of a message's origin); `subagent_id` is derived from its tail so the frontend can correlate it. */
get sessionId(): string {
return this.handle.sessionId;
}
/** Whether a round of the task is currently running. */
get running(): boolean {
return this.isRunning;
}
/** Terminal state of the most recent run; null if no round has ever completed. */
get exit(): SubagentExit | null {
return this.exitInfo;
}
/** Number of pending approval requests (the parent tool uses this to hint the model to poll again). */
get pendingApprovals(): number {
return this.approvals.length;
}
/** Whether there's unread output (buffered messages or text); used to re-check the predicate before waiting (see collect.ts). */
get hasPending(): boolean {
return this.messages.length > 0 || !this.textBuffer.isEmpty;
}
/**
* Starts a new round of the task on the child Session (async pump, doesn't block the caller).
* Throws if already disposed or still running (converted to an explanatory output by the
* caller).
*/
startRun(prompt: string): void {
if (this.killed) throw new Error("subagent session disposed");
if (this.isRunning) throw new Error("subagent is still running");
this.isRunning = true;
this.exitInfo = null;
void this.pump(prompt);
}
/** Takes the buffered child-session messages (already tagged with origin, for the parent tool to forward). */
drainMessages(): OmniMessage[] {
if (this.messages.length === 0) return [];
const out = this.messages;
this.messages = [];
return out;
}
/** Takes the currently unread child Agent text (including the drop marker); clears the buffer. */
drainText(): string {
return this.textBuffer.drain();
}
/** External wakeup (e.g. the parent tool call was aborted): makes a waiting `waitWake` return immediately. */
wakeup(): void {
this.wakeSignal.notify();
}
/** Waits for "woken up" or `ms` to expire, whichever comes first. */
async waitWake(ms: number): Promise<void> {
await this.wakeSignal.wait(ms);
}
/**
* Attaches an approval sink: the parent tool call active within the window hands in its own
* `ctx.approve`, and queued approval requests are consulted with Human through it one at a
* time. Returns a detach function (called when the window ends); a later attach replaces the
* former one.
*/
attachApprovalSink(approve: ApproveFn): () => void {
const epoch = ++this.sinkEpoch;
let onDetach!: () => void;
const detached = new Promise<void>((resolve) => {
onDetach = resolve;
});
this.sink = { approve, detached };
void this.pumpApprovals();
return () => {
if (this.sinkEpoch === epoch) this.sink = null;
onDetach();
};
}
/** Cleanup: aborts the current run, denies pending approvals, releases child Session resources; idempotent. */
kill(): void {
if (this.killed) return;
this.killed = true;
this.abortCtrl.abort();
for (const req of [...this.approvals]) this.settle(req, "deny");
// If running, released by pump's finally after it finishes; otherwise released immediately.
if (!this.isRunning) this.handle.dispose();
this.wakeSignal.notify();
}
/** Synchronous hard-kill path: the child Session runs in-process with no separate OS resources, so this is equivalent to `kill`. */
killHard(): void {
this.kill();
}
// -------------------------------------------------------------------------
// Internal: pump and buffering
// -------------------------------------------------------------------------
/** Drives one round of `handle.run`: buffers messages and text, settling the terminal state when it ends. */
private async pump(prompt: string): Promise<void> {
let wroteAny = false;
let childAbort: string | null = null;
try {
for await (const msg of this.handle.run({
prompt,
signal: this.abortCtrl.signal,
approve: this.childApprove,
})) {
this.bufferMessage(msg);
if ((msg.origin?.length ?? 0) === 1) {
const p = msg.payload as {
type?: string;
event_type?: string;
text?: string;
reason?: string;
};
// A direct child layer's abort event: the child session was interrupted/failed (LLM
// request error, user interruption, etc). A child session failure doesn't throw, it
// only emits an event, based on which this round is reported as failed rather than
// marked completed.
if (p.type === "abort") {
childAbort = p.reason ?? "aborted";
} else if (
p.type === "partial_text" &&
p.event_type === "delta" &&
typeof p.text === "string" &&
p.text
) {
wroteAny = true;
this.appendText(p.text);
}
}
this.wakeSignal.notify();
}
if (childAbort !== null) {
this.exitInfo = { status: "failed", note: `[subagent aborted: ${childAbort}]` };
} else if (!wroteAny) {
this.exitInfo = {
status: "completed",
note: "[subagent finished without a text answer]",
};
} else {
this.exitInfo = { status: "completed" };
}
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
this.exitInfo = { status: "failed", note: `[subagent error: ${message}]` };
} finally {
this.isRunning = false;
if (this.killed) this.handle.dispose();
this.wakeSignal.notify();
}
}
private bufferMessage(msg: OmniMessage): void {
this.messages.push(msg);
// Overflow drops the oldest: only affects frontend replay — the child Session's Trace and text buffer are unaffected.
if (this.messages.length > MESSAGE_BUFFER_CAP) this.messages.shift();
}
private appendText(text: string): void {
this.textBuffer.append(text);
}
// -------------------------------------------------------------------------
// Internal: approval queue
// -------------------------------------------------------------------------
/** Approval callback handed to the child Session: the request is queued and waits for some parent tool call to consult Human and give a decision. */
private readonly childApprove: ApproveFn = (toolCall) => {
if (this.killed) return Promise.resolve("deny");
return new Promise<ApprovalDecision>((resolve) => {
this.approvals.push({ toolCall, settled: false, resolve });
this.wakeSignal.notify(); // Wake the parent tool call waiting within the window, so it can consult as soon as possible
void this.pumpApprovals();
});
};
/** Settles an approval decision: first to arrive wins, late/duplicate decisions are ignored. */
private settle(req: PendingApproval, decision: ApprovalDecision): void {
if (req.settled) return;
req.settled = true;
const idx = this.approvals.indexOf(req);
if (idx >= 0) this.approvals.splice(idx, 1);
req.resolve(decision);
this.wakeSignal.notify();
}
/**
* Hands the request at the head of the queue to the currently attached approval sink, one at a
* time. Stops when the window ends (the sink is detached); unresolved requests stay queued for
* the next sink; if a consultation already in flight resolves late, the decision still takes
* effect via settle.
*/
private async pumpApprovals(): Promise<void> {
if (this.pumpingApprovals) return;
this.pumpingApprovals = true;
try {
while (this.sink && this.approvals.length > 0) {
const sink = this.sink;
const req = this.approvals[0]!;
const answer = sink.approve(req.toolCall).then(
(d) => this.settle(req, d),
() => this.settle(req, "deny"), // An approval sink error is treated as a denial (avoids leaving the child session stuck forever)
);
await Promise.race([answer, sink.detached]);
if (req.settled) continue;
if (this.sink && this.sink !== sink) continue; // The sink was replaced by a new call: retry with the new sink
break; // The sink was detached and still unresolved: stay queued for the next sink
}
} finally {
this.pumpingApprovals = false;
}
}
}
/** Converts a run's terminal state into a tool result (note is appended outside the truncation, so it isn't lost with long output). */
export function resultForSubagentExit(exit: SubagentExit | null): ToolResult {
if (!exit) return { stopReason: "completed" };
return { stopReason: exit.status, ...(exit.note !== undefined ? { note: exit.note } : {}) };
}
@@ -0,0 +1,79 @@
/**
* BuiltinTool abstraction — lets Environment avoid special-casing any specific tool name.
*
* Each builtin tool carries its own: `name`, the `definition` handed to the LLM, and a streaming
* `execute`. Environment dispatches purely by looking up `name`; an unknown tool collapses to an
* explanatory `tool_call_output`, never throwing. Adding a new tool later (e.g. file read/write,
* retrieval) only requires implementing this interface and registering it with the registry, with
* no changes needed to Environment.
*/
import type { OmniMessage, StopReason } from "../../omnimessage/index.js";
import type { ApproveFn, ToolDefinitionConfig } from "../../interfaces.js";
/**
* Tool execution context: runtime information needed to execute one tool call.
* Docs: /docs/tools § "Execution contract".
*/
export interface ToolExecutionContext {
/** Workspace absolute path; relative-path arguments should be resolved against it. */
workspaceDir: string;
/** The tool_call_id passed through unchanged, used to build streaming deltas and nested origin tags. */
toolCallId: string;
/** Abort signal; the tool should close out and return as soon as possible once it fires. */
signal?: AbortSignal;
/** The parent Agent's approval callback; run_subagent passes it through to the child Session so it inherits the parent's approval mode (unused by most tools). */
approve?: ApproveFn;
}
/**
* Tool execution result (the generator's return value); treated as `completed` if omitted.
* Docs: /docs/tools § "Execution contract".
*/
export interface ToolResult {
stopReason?: StopReason;
/**
* Terminal marker (e.g. `[exit code: 1]`): appended by Environment during its unified
* close-out, **outside** the maxOutputLength truncation, and streamed to the frontend as an
* extra chunk — so the failure marker isn't lost when long output gets truncated (it would be
* cut off if produced as a content delta instead).
*/
note?: string;
/**
* Images carried by the tool output (e.g. an image read by read_image): each entry is a
* `data:<mime>;base64,...` data URL. Attached by Environment during close-out: a single
* streaming delta carries it all at once before stop, plus the final complete
* `tool_call_output` (only carried on normal completion; images are not chunked and don't
* count toward text truncation).
*/
images?: string[];
}
/**
* Builtin tool interface. `execute` receives the already-parsed tool argument object and the
* execution context, streaming out OmniMessage as an async generator. Contract (a relaxed
* version — framing and close-out are handled uniformly by Environment):
*
* - **Own output**: yielding the **delta** of `partial_tool_call_output` is enough; `start`/`stop`
* are optional (Environment ignores the tool's start/stop and frames it itself), and there's
* **no need** to produce a complete `tool_call_output` either (the complete message,
* maxOutputLength forward truncation, and close-out are all derived by Environment from the
* deltas). If a tool does produce a complete `tool_call_output` anyway, Environment uses it as
* the basis for content and stop reason (tolerated for compatibility, not recommended).
* - **Nested forwarding**: yielding any message **tagged with origin** is passed through by
* Environment unchanged (e.g. run_subagent forwarding all of a child session's messages).
* - **Stop reason**: reported via the generator's return value (defaults to completed); a throw
* is collapsed by Environment into aborted/failed based on interruption/error, never
* propagating up as an exception.
* Docs: /docs/interfaces § "The inner tool contract: BuiltinTool"; /docs/tools § "Execution contract".
*/
export interface BuiltinTool {
/** Tool name (corresponds to the tool_call.name returned by the LLM). */
name: string;
/** Tool definition handed to the LLM (including description / parameters / permission / maxOutputLength). */
definition: ToolDefinitionConfig;
/** Executes one tool call: args is the already-parsed argument object, ctx is the runtime context. */
execute(
args: Record<string, unknown>,
ctx: ToolExecutionContext,
): AsyncGenerator<OmniMessage, ToolResult | void>;
}
+45
View File
@@ -0,0 +1,45 @@
/**
* @prismshadow/penguin-core — public entry point for the PenguinHarness core SDK.
*
* Exports the OmniMessage protocol, the three interface contracts (Human/LLM/Environment),
* and the runtime entry points for Agent / Session / context_engine along with their
* submodules (state / llm / environment / trace).
*
* Typical usage:
*
* ```ts
* const agent = await createAgent({ agentId: "default_agent" });
* const session = await agent.createSession({ workspaceDir, modelId });
* for await (const output of session.run([userText("...")])) { ... }
* ```
*/
// Protocol and interface contracts (foundation)
export * from "./omnimessage/index.js";
export * from "./interfaces.js";
// Submodules
export * from "./state/index.js";
export * from "./llm/index.js";
export * from "./environment/index.js";
export * from "./trace/index.js";
// Runtime entry points
export { ContextEngine } from "./engine/context-engine.js";
export type {
CompactAvailability,
CompactionSettings,
ContextEngineDeps,
EngineInitialState,
RunOptions,
TraceSink,
} from "./engine/context-engine.js";
export { Session } from "./session.js";
export type { SessionConfig } from "./session.js";
export { buildTitlePrompt, generateTitleWithLLM, sanitizeTitle } from "./session-title.js";
export type { SessionTitleResult } from "./session-title.js";
export { Agent, createAgent } from "./agent.js";
export type { CreateAgentOptions, CreateSessionOptions, ResumeSessionOptions } from "./agent.js";
/** SDK version number. */
export const VERSION = "0.0.1";
+279
View File
@@ -0,0 +1,279 @@
/**
* Internal SDK interface contracts: LLM, Environment.
*
* `context_engine` only handles OmniMessage; protocol conversion and concrete implementations
* are each interface's own responsibility.
* Human is not an "interface/class with methods" but the SDK's input/output boundary itself:
* output is streamed by `Session.run()` as an async generator, and input is delivered via
* `run`'s `RunOptions` — approvals are requested one at a time through the injected `approve`
* callback, and interruption goes through `signal`. Hence no Human interface is defined here.
*
* These types form the foundational contract shared by all units; implementing units integrate
* against them.
*
* Docs: packages/docs/content/interfaces.{zh,en}.md (site path /docs/interfaces) explains each
* contract and its extension seams — keep the page in sync when changing signatures here.
*/
import type {
ApprovalDecision,
OmniMessage,
StopReason,
ToolCallPayload,
ToolDefinition,
} from "./omnimessage/types.js";
// Concrete classes, used only for EnvironmentServices type annotations (type-only import; no runtime dependency, no circular reference).
import type { CommandSessionManager } from "./environment/tools/command/session-manager.js";
import type { SubagentSessionManager } from "./environment/tools/subagent/session-manager.js";
import type { ToolCallIdAllocator } from "./llm/tool-call-ids.js";
// ---------------------------------------------------------------------------
// Tool definitions and configuration
// ---------------------------------------------------------------------------
// ToolDefinition is defined in omnimessage/types.ts (session_meta embeds the full tool schema directly); re-exported here to keep the original import path.
export type { ToolDefinition } from "./omnimessage/types.js";
/** Tool permission: read-only / read-write. */
export type ToolPermission = "r" | "rw";
/**
* Runtime configuration for a single tool.
* Docs: /docs/tools § "Configuration fields".
*/
export interface ToolDefinitionConfig {
name: string;
description: string;
parameters?: Record<string, unknown>;
permission?: ToolPermission;
/**
* Which class of session model this entry targets: `"vision"` only for models that support
* images (e.g. read_image), `"text-only"` only for text-only models (e.g. describe_image);
* omitted means available for all models. Filtered by session model at assembly time
* (see `selectBuiltinToolsForModel`).
*/
forModel?: "vision" | "text-only";
/** Timeout for a single tool call (ms); on timeout, ends as `failed`; <=0 disables it. */
timeoutMs?: number;
/** Max length of tool output; Environment truncates from the front (keeping the head) if exceeded; <=0 disables it. */
maxOutputLength?: number;
}
export interface MCPServerConfig {
name: string;
config: Record<string, unknown>;
}
/** Set of tool configs required to initialize Environment. */
export interface ToolConfig {
customTools: ToolDefinitionConfig[];
mcpServers: MCPServerConfig[];
}
/**
* Per-tool approval callback: the Human boundary gives allow/deny for each complete `tool_call`.
* `context_engine` calls it once per tool call within a turn. Subagents forward the parent's
* approval callback, so the child Agent **inherits the parent Agent's approval mode**.
* Docs: /docs/interfaces § "ApproveFn".
*/
export type ApproveFn = (toolCall: OmniMessage<ToolCallPayload>) => Promise<ApprovalDecision>;
// ---------------------------------------------------------------------------
// LLM interface
// ---------------------------------------------------------------------------
export type ThinkingLevelName = "none" | "low" | "medium" | "high" | "xhigh";
/**
* GenerativeModel initialization config.
* Docs: /docs/interfaces § "GenerativeModelConfig".
*/
export interface GenerativeModelConfig {
modelId: string;
apiKey?: string;
baseUrl?: string;
/**
* AgentHub client protocol (`openai` / `claude-4-8` / `deepseek-v4` / …). If omitted, AgentHub
* infers it from `modelId`; custom-named models or third-party models using the OpenAI protocol
* must specify it explicitly.
*/
clientType?: string;
tools: ToolDefinition[];
/** Full system Prompt after placeholder substitution in the system_config.system_prompt template. */
systemPrompt?: string;
contextWindow?: number;
maxTokens?: number;
thinkingLevel?: ThinkingLevelName;
/** LLM Request timeout (ms): from system_config.model.timeoutMs; <=0 disables it. Defaults to 120000. */
requestTimeoutMs?: number;
/**
* tool_call_id uniqueness registry (Session-level). Pass the same instance when rebuilding a new
* GenerativeModel on compaction so the uniqueness scope covers the whole Session; defaults to a fresh
* one. See llm/tool-call-ids.ts.
*/
toolCallIds?: ToolCallIdAllocator;
}
export interface GenerativeModelParameters {
/** OmniMessage array for the input newly added this turn; implementations must merge it into a single UniMessage (multiple roles not accepted). */
newMessages: OmniMessage[];
signal?: AbortSignal;
}
/**
* The terminal state of an LLM request, returned as the **return value** of the `streamGenerate`
* async generator (not a yielded message). The status values share the same five-value protocol
* as OmniMessage `stop_reason`:
* - `completed`: finished normally (already produced `token_usage`);
* - `timeout`: LLM timed out or lost connection, needs reconnect — retried by `context_engine`
* within the same run;
* - `malformed`: AgentHub response failed JSON parsing, needs reconnect — also retried by
* `context_engine`;
* - `aborted`: user-initiated interruption — stop and hand back to the user;
* - `failed`: other non-retryable errors (auth/params, etc.) — stop and hand back to the user
* (`message` provides the display text).
* Docs: /docs/interfaces § "LLMOutcome semantics".
*/
export interface LLMOutcome {
status: StopReason;
message?: string;
}
/**
* A stateful LLM object attached to a Session.
* `streamGenerate` yields streaming `partial_*` messages as an async generator, and appends the
* corresponding complete `model_msg` once each fragment ends; Token usage is emitted as a
* `token_usage` event_msg. **Never throws to `context_engine`**: any interruption/exception is
* closed off in well-formed structure and returned normally, and **must** report the terminal
* state via `LLMOutcome` — error handling happens entirely inside the LLM interface, and
* `context_engine` only decides subsequent actions based on the outcome.
* Docs: /docs/interfaces § "LLMInterface".
*/
export interface LLMInterface {
streamGenerate(parameters: GenerativeModelParameters): AsyncGenerator<OmniMessage, LLMOutcome>;
}
// ---------------------------------------------------------------------------
// Environment interface
// ---------------------------------------------------------------------------
/**
* Handle for a child Agent session: derived by `SubagentRunner.spawn`,
* representing a child Session that can run over multiple turns. Deriving (spawn) is separate
* from running (run), so the same child Session can accept an additional Prompt and keep running
* after a turn ends (a long-running subagent, accessed via `input_subagent`).
* Docs: /docs/interfaces § "Subagent interfaces".
*/
export interface SubagentHandle {
/** The child Session's id: the origin hop of messages produced by run; `subagent_id` is derived from its tail for the frontend to correlate. */
sessionId: string;
/**
* Runs one turn of a task on the child Session. Emitted child-session messages **all already
* carry the origin marker** (the child Session id); the first message of the first run is the
* child Session's `session_meta`, and tool_calls received by the forwarded approval callback
* carry origin as well.
*/
run(input: {
/** The task Prompt handed to the child Agent. */
prompt: string;
signal?: AbortSignal;
/** The parent Agent's approval callback; forwarded to the child Session to inherit the parent's approval mode. */
approve?: ApproveFn;
}): AsyncGenerator<OmniMessage>;
/** Releases runtime resources held by the child Session (e.g. its managed command sessions). Idempotent. */
dispose(): void;
}
/**
* Child Agent runner: injected into the `run_subagent` tool so it can
* derive and run a child Agent without a reverse dependency on Agent/Session, avoiding circular
* dependencies. The concrete implementation is provided by the SDK composition layer (where
* `createAgent` lives), which internally derives via `createAgent` → `createSession` and hands
* back a `SubagentHandle`.
* Docs: /docs/interfaces § "Subagent interfaces".
*/
export interface SubagentRunner {
/**
* Derives a child Agent and creates a child Session. Precheck errors such as exceeding the
* depth limit or a nonexistent target agent are expressed by throwing (collapsed to `failed`
* by Environment).
*/
spawn(input: {
/** The child Agent's agentId; if omitted, reuses the current Agent (self-invocation). */
agentId?: string;
/** The Model used by the child Session; if omitted, uses the Project's default Model. */
modelId?: string;
}): Promise<SubagentHandle>;
}
/**
* Proxy-reading service for describe_image: injected when the session model doesn't support
* images (vision=false) — images are handed to the configured vision model for description and
* the tool returns text, avoiding a 400 from feeding images back into a tool_result for a
* provider that doesn't support images.
* Docs: /docs/interfaces § "VisionDescriberService".
*/
export interface VisionDescriberService {
/** Vision model id; null when the Project has no `vision_model` configured (or it's invalid), in which case the tool ends with a failed explanation. */
modelId: string | null;
/** Constructs a single-shot LLM for this vision model (no tools, no system prompt); omitted when `modelId` is null. */
createLLM?: () => LLMInterface;
}
/**
* Runtime services Environment injects into individual tools (e.g. `run_subagent` needs `SubagentRunner`); most tools don't use these.
* Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig".
*/
export interface EnvironmentServices {
subagentRunner?: SubagentRunner;
/** Injected when the session model doesn't support images: for describe_image's single-shot vision-model proxy reading. */
visionDescriber?: VisionDescriberService;
/** Registry of long-running command sessions (shared by `exec_command` / `input_command`); constructed and injected internally by Environment. */
commandSessions?: CommandSessionManager;
/** Registry of background subagent sessions (shared by `run_subagent` / `input_subagent`); constructed and injected internally by Environment. */
subagentSessions?: SubagentSessionManager;
}
/** Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig". */
export interface EnvironmentConfig {
workspaceDir: string;
toolConfig: ToolConfig;
/** Runtime services (optional); Environment forwards these to each tool factory to use as needed. */
services?: EnvironmentServices;
/**
* Agent vault environment variables (key-value pairs, taken from the Agent's
* `agent_state/.vault.toml`): injected into the exec_command / input_command subprocess
* environment; hardened entries cannot be overridden.
*/
vault?: Record<string, string>;
}
/**
* An approved tool-call execution request.
* Docs: /docs/interfaces § "ToolExecutionRequest and EnvironmentConfig".
*/
export interface ToolExecutionRequest {
/** The OmniMessage whose payload.type === "tool_call". */
toolCall: OmniMessage<ToolCallPayload>;
signal?: AbortSignal;
/** The parent Agent's approval callback; forwarded to tools that need to derive a child Session (run_subagent), implementing approval inheritance. */
approve?: ApproveFn;
}
/**
* Environment interface: executes approved tool calls within the Workspace.
* `executeTool` yields `partial_tool_call_output` as an async generator and ends with exactly one
* complete `tool_call_output`; nested session messages carrying an origin marker (e.g. forwarded
* by run_subagent) pass through unchanged.
*
* **Rendering** of tool calls is not this interface's concern (nor core's): streaming rendering is
* handled by the CLI / Web frontend itself.
* Docs: /docs/interfaces § "EnvironmentInterface".
*/
export interface EnvironmentInterface {
listTools(): Promise<ToolDefinition[]>;
executeTool(request: ToolExecutionRequest): AsyncGenerator<OmniMessage>;
/** Looks up a tool's permission level (for frontend permission-mode decisions); returns undefined for unknown tools. */
toolPermission(name: string): ToolPermission | undefined;
/** Releases runtime resources held by the environment (e.g. managed long-running command sessions); called by the host when the Session ends. Optional, idempotent. */
dispose?(): void;
}
+9
View File
@@ -0,0 +1,9 @@
/** Local-timezone date formatting (internal shared helper, not exported via the barrel). */
/** Format a date as local `yyyy-mm-dd` (local timezone, 4-digit year, zero-padded 2-digit month/day). */
export function formatLocalDate(date: Date): string {
const year = date.getFullYear().toString().padStart(4, "0");
const month = (date.getMonth() + 1).toString().padStart(2, "0");
const day = date.getDate().toString().padStart(2, "0");
return `${year}-${month}-${day}`;
}
@@ -0,0 +1,171 @@
/**
* Session creation helpers (used by `agent.createSession` for assembly, not exported
* via the barrel): Session id generation, runtime environment fields, and temp
* Workspace creation.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { randomBytes, randomUUID } from "node:crypto";
import { formatLocalDate } from "./dates.js";
import type { SessionEnvironmentValues } from "../state/agent-state.js";
import { workspacesDir } from "../state/index.js";
import { userText } from "../omnimessage/index.js";
import type { OmniMessage } from "../omnimessage/index.js";
/** Session runtime environment fields: the placeholder substitution values for `assembleSystemPrompt`; producer and consumer share the same type. */
export type SessionEnvironment = SessionEnvironmentValues;
/** Generate a Session id of the form `session-YYYY-MM-DD-HH-mm-ss-<8-hex>` (local timezone, zero-padded: 4-digit year, 2 digits for the rest; hex from randomUUID). */
export function formatSessionId(date: Date = new Date()): string {
const pad = (n: number) => n.toString().padStart(2, "0");
const ts =
`${formatLocalDate(date)}` +
`-${pad(date.getHours())}-${pad(date.getMinutes())}-${pad(date.getSeconds())}`;
const hex = randomUUID().replace(/-/g, "").slice(0, 8);
return `session-${ts}-${hex}`;
}
/**
* Generate this Session's runtime environment fields (injected via specific
* placeholders in the system prompt).
* This is system-generated runtime context, not sourced from Agent State / Workspace files.
*/
export function sessionEnvironment(
workspaceDir: string,
sessionId: string,
ids: { agentId: string; projectDir: string },
date = new Date(),
): SessionEnvironment {
return {
sessionId,
cwd: workspaceDir,
agentId: ids.agentId,
projectDir: ids.projectDir,
platform: process.platform,
osVersion: getOsVersion(),
date: formatLocalDate(date),
};
}
function getOsVersion(): string {
// os.* is a stable built-in API that normally doesn't throw; but this function only
// builds a single line of environment info for the system prompt, so it's not worth
// letting an exception take down createSession — fall back to "unknown" instead.
try {
if (process.platform === "win32") {
return `${os.version()} ${os.release()}`;
}
return `${os.type()} ${os.release()}`;
} catch {
return "unknown";
}
}
/** The 8-hex space is 2^32, so the odds of consecutive collisions are negligible; the cap only guards against an infinite loop caused by an abnormal filesystem. */
const MAX_TMP_ID_ATTEMPTS = 16;
/**
* Create a temporary Workspace under `<agent>/workspaces/<workspace_id>`, where the
* directory name is the workspace_id, shaped like `tmp-<8hex>`; if it collides with
* an existing directory, regenerate the id. No symlinks are created inside the Workspace:
* the model composes absolute paths (to Agent State, scratchpad, etc.) directly from the
* Environment placeholders (Project Dir / Agent ID) in the system prompt.
*/
export async function createTempWorkspace(
root: string,
projectId: string,
agentId: string,
): Promise<string> {
const base = workspacesDir(root, projectId, agentId);
await fs.mkdir(base, { recursive: true });
// The final directory must use a non-recursive mkdir: recursive mkdir succeeds
// silently when the directory already exists, which would put a new Session into
// an existing temp Workspace; EEXIST means an id collision, so retry with a new id.
for (let attempt = 0; attempt < MAX_TMP_ID_ATTEMPTS; attempt++) {
const dir = path.join(base, `tmp-${randomUUID().slice(0, 8)}`);
try {
await fs.mkdir(dir);
return dir;
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
}
}
throw new Error(
`failed to allocate a unique temp workspace id under ${base} after ${MAX_TMP_ID_ATTEMPTS} attempts`,
);
}
/** Maps a data URL's mime type to a file extension on disk; unknown mimes use bin (the image-reading tool sniffs the magic bytes and doesn't rely on the extension). */
const MIME_TO_EXT: Record<string, string> = {
"image/png": "png",
"image/jpeg": "jpg",
"image/gif": "gif",
"image/webp": "webp",
};
/**
* Input conversion for when the session model doesn't support images: image messages
* in the `run` input are written to disk as files (base64 data URLs are saved to the
* session scratchpad; http(s) URLs are referenced
* as-is), and the path/URL is appended to the user text (an `[attached image: …]`
* line); the image message itself is removed from the input — the model views it by
* path via describe_image (read on its behalf by a vision model), and images never
* enter that session's history directly.
* Returns the input unchanged when there are no images; an image that can't be
* parsed is replaced with an explanatory line rather than silently dropped.
*/
export async function imagesToScratchpadPaths(
input: OmniMessage[],
dir: string,
): Promise<OmniMessage[]> {
const isImage = (m: OmniMessage): boolean =>
(m.payload as { type?: string }).type === "image_url";
if (!input.some(isImage)) return input;
const lines: string[] = [];
for (const msg of input) {
if (!isImage(msg)) continue;
const url = (msg.payload as { image_url?: string }).image_url ?? "";
if (/^https?:\/\//i.test(url)) {
lines.push(`[attached image: ${url}]`);
continue;
}
const match = /^data:([^;,]+);base64,(.+)$/s.exec(url);
if (!match) {
lines.push("[an attached image could not be saved and was dropped]");
continue;
}
await fs.mkdir(dir, { recursive: true });
const ext = MIME_TO_EXT[match[1]!.toLowerCase()] ?? "bin";
// Filename = upload-<8 random hex chars> (same convention as project-<8hex>; the
// prefix distinguishes model-generated temp files).
// "wx" flag does exclusive creation to avoid name collisions: on the rare chance of a collision, retry with a new random value.
let file: string;
for (;;) {
file = path.join(dir, `upload-${randomBytes(4).toString("hex")}.${ext}`);
try {
await fs.writeFile(file, Buffer.from(match[2]!, "base64"), { flag: "wx" });
break;
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "EEXIST") throw err;
}
}
lines.push(`[attached image: ${file}]`);
}
// Concatenation: the path lines are appended after the last user text message; if the input is images only, add a plain path-only text message.
const rest = input.filter((m) => !isImage(m));
const suffix = lines.join("\n");
const lastTextIdx = rest.findLastIndex((m) => {
const p = m.payload as { type?: string; role?: string };
return p.type === "text" && p.role === "user";
});
if (lastTextIdx === -1) return [...rest, userText(suffix)];
return rest.map((m, i) => {
if (i !== lastTextIdx) return m;
const p = m.payload as { type: string; role: string; text: string };
return { ...m, payload: { ...p, text: `${p.text}\n\n${suffix}` } } as OmniMessage;
});
}
File diff suppressed because it is too large Load Diff
+22
View File
@@ -0,0 +1,22 @@
/**
* LLM interface module entry point.
*
* Exports `GenerativeModel` (the LLMInterface implementation), along with internal
* pure conversion functions for unit testing (message merging, event translation,
* token accounting, UniConfig construction, retry determination).
*/
export {
GenerativeModel,
EventTranslator,
groupHistoryToUniMessages,
mergeOmniToUniMessage,
translateEvents,
usageToTokenCounts,
isMalformedJsonParseError,
isIncompleteStreamError,
isRetryableError,
mapThinkingLevel,
toolDefinitionsToSchemas,
buildUniConfig,
} from "./generative-model.js";
export { ToolCallIdAllocator, stripToolCallIdSuffix } from "./tool-call-ids.js";
+51
View File
@@ -0,0 +1,51 @@
/**
* Session-level uniqueness for tool_call_id.
*
* Some providers don't produce a real call id: e.g. Gemini's functionCall has no id, so AgentHub
* uses the **function name** as the `tool_call_id` — consecutive/parallel calls to the same tool then
* all share one id. But the OmniMessage world (engine dispatch/pairing, approval routing, frontend
* tool-card attribution) keys on `tool_call_id`, and a collision lets a later call overwrite the
* earlier one (parallel same-name calls in one turn can even be dropped entirely).
*
* Approach: inbound, `EventTranslator` disambiguates duplicate ids with a `#n` suffix (the first keeps
* the original id); outbound (returning tool_result, replaying history on resume) uses
* `stripToolCallIdSuffix` to strip the suffix and restore the original — Gemini's functionResponse
* pairs by using `tool_call_id` as the name, so it must be restored to the function name. The registry
* lives at Session level (the new GenerativeModel rebuilt on compaction shares the same instance), and
* on resume `setHistory` seeds it with historical ids, so the uniqueness scope covers the entire
* context the frontend renders.
* Docs: /docs/interfaces § "The built-in implementation: GenerativeModel".
*/
export class ToolCallIdAllocator {
/** OmniMessage-level tool_call_ids already taken in this Session (history-seeded + allocated). */
private used = new Set<string>();
/** Register an already-used id (for resume seeding); registering twice is harmless. */
markUsed(id: string): void {
this.used.add(id);
}
/**
* Allocate a Session-unique id for a provider-reported tool_call_id: if unused, keep the original;
* if already used (a repeat call from a name-as-id provider), take the first free `origId#n` (n from 2).
* Providers with truly unique ids (OpenAI `call_*` / Claude `toolu_*`) never collide, so they pass through unchanged.
*/
allocate(providerId: string): string {
let id = providerId;
for (let n = 2; this.used.has(id); n += 1) {
id = `${providerId}#${n}`;
}
this.used.add(id);
return id;
}
}
/**
* Strip the `#n` suffix added by `allocate`, restoring the provider's original id (returns as-is when
* there's no suffix; idempotent). On resume there's no registry to compare against, so it trims by
* shape: real ids from known providers (OpenAI/Claude `call_*`/`toolu_*`, Gemini function names — `#`
* isn't a valid function-name char) never end in `#<digits>`, so they aren't harmed.
*/
export function stripToolCallIdSuffix(id: string): string {
return id.replace(/#\d+$/, "");
}
+155
View File
@@ -0,0 +1,155 @@
/**
* Aggregates streaming partial_* messages into a complete model_msg.
*
* When recording a Trace, streaming `partial_*` messages must first be joined into a complete
* `model_msg` before writing. This module provides:
* - `PartialAggregator`: a stateful aggregator, pushed one message at a time, producing a
* complete message when a fragment ends with `stop`;
* - `aggregateAll`: a one-shot pass that collapses `partial_*` messages in an array into
* complete messages.
*
* Complete / event / session_meta messages pass through unchanged, preserving their original
* order.
* Docs: /docs/omni-message § "The streaming discipline".
*/
import { assistantText, thinkingMessage, toolCall, toolCallOutput } from "./builders.js";
import type { OmniMessage, PartialModelPayload, StopReason } from "./types.js";
import { isPartialPayload } from "./types.js";
type PartialKind = PartialModelPayload["type"];
interface OpenFragment {
kind: PartialKind;
/** Accumulation buffer for text / thinking / tool_call arguments / tool_call_output. */
buffer: string;
name?: string;
toolCallId?: string;
/** Images carried by tool_call_output (images aren't incremental — a single delta carries the whole set; a later one overwrites). */
images?: string[];
lastStopReason: StopReason;
}
/** Merge key for partial fragments: same type + same tool_call_id counts as the same fragment. */
function fragmentKey(p: PartialModelPayload): string {
const id = "tool_call_id" in p ? p.tool_call_id : "";
return `${p.type}::${id}`;
}
function finalize(frag: OpenFragment): OmniMessage {
switch (frag.kind) {
case "partial_text":
return assistantText(frag.buffer, frag.lastStopReason);
case "partial_thinking":
return thinkingMessage(frag.buffer, frag.lastStopReason);
case "partial_tool_call":
return toolCall({
name: frag.name ?? "",
arguments: frag.buffer,
toolCallId: frag.toolCallId ?? "",
stopReason: frag.lastStopReason,
});
case "partial_tool_call_output":
return toolCallOutput({
output: frag.buffer,
toolCallId: frag.toolCallId ?? "",
stopReason: frag.lastStopReason,
...(frag.images !== undefined ? { images: frag.images } : {}),
});
}
}
function appendDelta(frag: OpenFragment, p: PartialModelPayload): void {
switch (p.type) {
case "partial_text":
frag.buffer += p.text;
break;
case "partial_thinking":
frag.buffer += p.thinking;
break;
case "partial_tool_call":
frag.buffer += p.arguments;
if (p.name) frag.name = p.name;
frag.toolCallId = p.tool_call_id;
break;
case "partial_tool_call_output":
frag.buffer += p.output;
if (p.images && p.images.length > 0) frag.images = p.images;
frag.toolCallId = p.tool_call_id;
break;
}
if (p.stop_reason !== undefined) frag.lastStopReason = p.stop_reason;
}
function newFragment(p: PartialModelPayload): OpenFragment {
const frag: OpenFragment = {
kind: p.type,
buffer: "",
lastStopReason: "completed",
};
if (p.type === "partial_tool_call") {
frag.name = p.name;
frag.toolCallId = p.tool_call_id;
} else if (p.type === "partial_tool_call_output") {
frag.toolCallId = p.tool_call_id;
}
return frag;
}
/**
* Stateful aggregator. Pushed one message at a time via `push`:
* - complete / event / session_meta messages are returned unchanged;
* - `partial_*` messages accumulate into an internal fragment, producing a complete message
* when `event_type === "stop"`;
* - `flush` forcibly emits any fragments that haven't yet received a stop.
*/
export class PartialAggregator {
private open = new Map<string, OpenFragment>();
push(msg: OmniMessage): OmniMessage[] {
if (!isPartialPayload(msg.payload)) {
return [msg];
}
const p = msg.payload;
const key = fragmentKey(p);
let frag = this.open.get(key);
if (p.event_type === "start") {
// start reopens a fragment; if a fragment with the same key already exists (out-of-order), finalize it first.
const out: OmniMessage[] = [];
if (frag) out.push(finalize(frag));
frag = newFragment(p);
appendDelta(frag, p);
this.open.set(key, frag);
return out;
}
if (!frag) {
// delta/stop without a preceding start: handle leniently, creating a new fragment as needed.
frag = newFragment(p);
this.open.set(key, frag);
}
appendDelta(frag, p);
if (p.event_type === "stop") {
this.open.delete(key);
return [finalize(frag)];
}
return [];
}
/** Finalizes: emits all still-open fragments (in the order they were opened). */
flush(): OmniMessage[] {
const out = [...this.open.values()].map(finalize);
this.open.clear();
return out;
}
}
/** One-shot aggregation: keeps non-partial messages in their original order, collapsing partial ones into complete messages. */
export function aggregateAll(messages: OmniMessage[]): OmniMessage[] {
const agg = new PartialAggregator();
const out: OmniMessage[] = [];
for (const msg of messages) out.push(...agg.push(msg));
out.push(...agg.flush());
return out;
}
+342
View File
@@ -0,0 +1,342 @@
/**
* OmniMessage builders. All modules create messages exclusively through these builders, avoiding
* ad hoc protocol structures scattered across the codebase.
* Every builder writes an ISO 8601 UTC timestamp.
* Docs: /docs/omni-message § "Builders and guards".
*/
import type {
AbortPayload,
ApprovalDecision,
ApprovalDecisionPayload,
CompactionBeginPayload,
CompactionEndPayload,
CompactionMode,
CompactionReason,
EventMessage,
ImageUrlPayload,
InlineDataPayload,
InlineThinkingPayload,
MessageOrigin,
ModelMessage,
OmniMessage,
PartialTextPayload,
PartialThinkingPayload,
PartialToolCallOutputPayload,
PartialToolCallPayload,
RequestBeginPayload,
RequestEndPayload,
Role,
SessionMetaMessage,
SessionMetaPayload,
StopReason,
StreamEventType,
SubagentPayload,
TextPayload,
ThinkingPayload,
TokenCounts,
TokenUsagePayload,
ToolCallOutputPayload,
ToolCallPayload,
} from "./types.js";
/** The current moment's ISO 8601 UTC timestamp. */
function nowIso(): string {
return new Date().toISOString();
}
function model<P extends ModelMessage["payload"]>(payload: P): OmniMessage<P> {
return { timestamp: nowIso(), type: "model_msg", payload };
}
function event<P extends EventMessage["payload"]>(payload: P): OmniMessage<P> {
return { timestamp: nowIso(), type: "event_msg", payload };
}
// session_meta ---------------------------------------------------------------
export function sessionMeta(payload: SessionMetaPayload): SessionMetaMessage {
return { timestamp: nowIso(), type: "session_meta", payload };
}
// Complete model_msg -----------------------------------------------------------
/**
* Provider fidelity fields: kept as-is and restored verbatim on replay.
* Builder convention: positional-argument-style builders carry these in a trailing `fidelity`
* object (narrowed via Pick per payload type — e.g. thinking only has signature); object-argument-
* style builders (toolCall) flatten `fidelity` fields into the parameter object alongside
* `stopReason`, mirroring the payload structure directly.
*/
export interface FidelityFields {
phase?: string | null;
signature?: string;
}
export function textMessage(
role: Role,
text: string,
stopReason: StopReason = "completed",
fidelity?: FidelityFields,
): OmniMessage<TextPayload> {
return model({
type: "text",
role,
text,
stop_reason: stopReason,
...(fidelity?.phase != null ? { phase: fidelity.phase } : {}),
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export const userText = (text: string): OmniMessage<TextPayload> => textMessage("user", text);
export const assistantText = (
text: string,
stopReason: StopReason = "completed",
fidelity?: FidelityFields,
): OmniMessage<TextPayload> => textMessage("assistant", text, stopReason, fidelity);
export function imageUrlMessage(imageUrl: string): OmniMessage<ImageUrlPayload> {
return model({
type: "image_url",
role: "user",
image_url: imageUrl,
stop_reason: "completed",
});
}
export function inlineData(
role: Role,
data: string,
mimeType: string,
fidelity?: Pick<FidelityFields, "signature">,
): OmniMessage<InlineDataPayload> {
return model({
type: "inline_data",
role,
data,
mime_type: mimeType,
stop_reason: "completed",
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export function thinkingMessage(
thinking: string,
stopReason: StopReason = "completed",
fidelity?: Pick<FidelityFields, "signature">,
): OmniMessage<ThinkingPayload> {
return model({
type: "thinking",
role: "assistant",
thinking,
stop_reason: stopReason,
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export function inlineThinking(
data: string,
mimeType: string,
fidelity?: Pick<FidelityFields, "signature">,
): OmniMessage<InlineThinkingPayload> {
return model({
type: "inline_thinking",
role: "assistant",
data,
mime_type: mimeType,
stop_reason: "completed",
...(fidelity?.signature !== undefined ? { signature: fidelity.signature } : {}),
});
}
export function toolCall(args: {
name: string;
arguments: string;
toolCallId: string;
stopReason?: StopReason;
signature?: string;
}): OmniMessage<ToolCallPayload> {
return model({
type: "tool_call",
role: "assistant",
name: args.name,
arguments: args.arguments,
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
...(args.signature !== undefined ? { signature: args.signature } : {}),
});
}
export function toolCallOutput(args: {
output: string;
toolCallId: string;
stopReason?: StopReason;
/** Images carried by the tool output (array of data URLs); images aren't incremental — a single delta carries the whole set in the streaming path, and the complete message carries them too. */
images?: string[];
}): OmniMessage<ToolCallOutputPayload> {
return model({
type: "tool_call_output",
role: "user",
output: args.output,
...(args.images !== undefined && args.images.length > 0 ? { images: args.images } : {}),
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
});
}
// Streaming partial_* model_msg -------------------------------------------------
export function partialText(
eventType: StreamEventType,
text = "",
stopReason: StopReason = "completed",
): OmniMessage<PartialTextPayload> {
return model({
type: "partial_text",
role: "assistant",
event_type: eventType,
text,
stop_reason: stopReason,
});
}
export function partialThinking(
eventType: StreamEventType,
thinking = "",
stopReason: StopReason = "completed",
): OmniMessage<PartialThinkingPayload> {
return model({
type: "partial_thinking",
role: "assistant",
event_type: eventType,
thinking,
stop_reason: stopReason,
});
}
export function partialToolCall(args: {
eventType: StreamEventType;
name: string;
arguments?: string;
toolCallId: string;
stopReason?: StopReason;
}): OmniMessage<PartialToolCallPayload> {
return model({
type: "partial_tool_call",
role: "assistant",
event_type: args.eventType,
name: args.name,
arguments: args.arguments ?? "",
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
});
}
export function partialToolCallOutput(args: {
eventType: StreamEventType;
output?: string;
toolCallId: string;
stopReason?: StopReason;
/** Images carried by the tool output (array of data URLs); images aren't incremental — a single delta carries the whole set. */
images?: string[];
}): OmniMessage<PartialToolCallOutputPayload> {
return model({
type: "partial_tool_call_output",
role: "user",
event_type: args.eventType,
output: args.output ?? "",
...(args.images !== undefined && args.images.length > 0 ? { images: args.images } : {}),
tool_call_id: args.toolCallId,
stop_reason: args.stopReason ?? "completed",
});
}
// event_msg -------------------------------------------------------------------
export function approvalDecision(
decision: ApprovalDecision,
toolCallId: string,
): OmniMessage<ApprovalDecisionPayload> {
return event({ type: "approval_decision", decision, tool_call_id: toolCallId });
}
export function abortEvent(reason: string | null = null): OmniMessage<AbortPayload> {
return event({ type: "abort", reason });
}
/** request begin event: marks the start of one LLM Request. */
export function requestBegin(): OmniMessage<RequestBeginPayload> {
return event({ type: "request_begin" });
}
/** request end event: carries the terminal state (`completed` means this turn was already committed to AgentHub). */
export function requestEnd(status: StopReason): OmniMessage<RequestEndPayload> {
return event({ type: "request_end", status });
}
/** compaction begin event: carries the trigger reason, mode, current context usage, and cumulative Session turn count. */
export function compactionBegin(args: {
reason: CompactionReason;
mode: CompactionMode;
context: number;
turns: number;
}): OmniMessage<CompactionBeginPayload> {
return event({
type: "compaction_begin",
reason: args.reason,
mode: args.mode,
context: args.context,
turns: args.turns,
});
}
/** compaction end event: carries the compaction result (non-`completed` means compaction was abandoned and the original context is kept). */
export function compactionEnd(args: {
reason: CompactionReason;
mode: CompactionMode;
status: StopReason;
}): OmniMessage<CompactionEndPayload> {
return event({
type: "compaction_end",
reason: args.reason,
mode: args.mode,
status: args.status,
});
}
/** subagent derivation pointer event: records only the direct child session's Session id (written to the parent Trace by context_engine). */
export function subagentEvent(sessionId: string): OmniMessage<SubagentPayload> {
return event({ type: "subagent", session_id: sessionId });
}
export function emptyTokenCounts(): TokenCounts {
return { cache_read: 0, cache_write: 0, output: 0, total: 0 };
}
export function tokenUsage(
session: TokenCounts,
request: TokenCounts,
): OmniMessage<TokenUsagePayload> {
return event({ type: "token_usage", session, request });
}
/** Adds two sets of Token counts together, used to maintain cumulative Session usage. */
export function addTokenCounts(a: TokenCounts, b: TokenCounts): TokenCounts {
return {
cache_read: a.cache_read + b.cache_read,
cache_write: a.cache_write + b.cache_write,
output: a.output + b.output,
total: a.total + b.total,
};
}
/**
* Marks a message with a nested-origin tag: prepends one hop (a child Session id) to the front
* of `origin`, outer-to-inner.
* Used by host tools (e.g. run_subagent) when forwarding child-session messages; an absent
* `origin` means the message comes from the main Session.
*/
export function withOrigin<M extends OmniMessage>(msg: M, sessionId: MessageOrigin): M {
return { ...msg, origin: [sessionId, ...(msg.origin ?? [])] };
}
+3
View File
@@ -0,0 +1,3 @@
export * from "./types.js";
export * from "./builders.js";
export * from "./aggregate.js";
+400
View File
@@ -0,0 +1,400 @@
/**
* OmniMessage — PenguinHarness's primary message protocol.
*
* All messages share one envelope: `timestamp` (ISO 8601 UTC), `type`, and `payload`.
* The outer `type` falls into three categories:
* - `session_meta`: Session metadata;
* - `model_msg`: model input/output messages (both complete messages and streaming
* `partial_*` messages);
* - `event_msg`: control/statistics events during execution.
*
* Trace records only: `session_meta`, complete `model_msg`, and all `event_msg`;
* the Human interface communicates using: complete `model_msg`, streaming `partial_*`, and all
* `event_msg`.
*
* Docs: packages/docs/content/omni-message.{zh,en}.md (site path /docs/omni-message) documents
* this protocol payload-for-payload — keep the page in sync when changing types here.
*/
/** The outer message category. */
export type OmniMessageType = "session_meta" | "model_msg" | "event_msg";
/** The message's originating role. */
export type Role = "user" | "assistant";
/**
* The reason a model response or message generation ended. Only five protocol values are
* allowed:
* - `completed`: finished normally, including completed text, thinking, tool requests, or
* tool output;
* - `failed`: a non-retryable error or tool execution failure;
* - `aborted`: user-initiated interruption or cancellation;
* - `timeout`: LLM request timed out;
* - `malformed`: the LLM response was malformed (e.g. AgentHub JSON parsing exception).
* Only LLM timeout / malformed trigger a context_engine reconnect.
* Docs: /docs/omni-message § "stop_reason".
*/
export type StopReason = "completed" | "failed" | "aborted" | "timeout" | "malformed";
/** The event phase of a streaming fragment. `stop` marks the end of a fragment and usually carries no incremental content. */
export type StreamEventType = "start" | "delta" | "stop";
/**
* Nested-origin marker: a child Session id. The message envelope's `origin` is a chain of child
* Session ids ordered **outer-to-inner**, identifying that the message comes from a nested child
* session (e.g. a child Session derived by `run_subagent`); each layer of host-tool forwarding
* prepends one more hop at the front. **An absent `origin` (the message carries no `origin`)
* means the message comes from the main Session itself** (an empty array is never produced
* either). Only session_id is recorded: the corresponding tool_call / agent info can be obtained
* from the `run_subagent` tool_call in the parent session's stream and the child Session's own
* Trace (session_meta).
* Docs: /docs/omni-message § "origin: the Subagent chain".
*/
export type MessageOrigin = string;
/** The approval decision for a tool call. */
export type ApprovalDecision = "allow" | "deny";
/** Token counts (input/output/cache/total). */
export interface TokenCounts {
cache_read: number;
cache_write: number;
output: number;
total: number;
}
// ---------------------------------------------------------------------------
// session_meta
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "session_meta"
/** Tool definition passed to the LLM (OpenAI/JSON Schema style). */
export interface ToolDefinition {
name: string;
description: string;
parameters?: Record<string, unknown>;
}
export interface SessionMetaPayload {
session_id: string;
/** The session model's provider group (paired with `model_id` to form a model reference). */
provider: string;
/** The session model's upstream model_id (the request id sent to AgentHub; paired with `provider`). */
model_id: string;
model_context_window: number | string;
/** The system prompt actually used by this Session (the assembled result with environment placeholders already substituted). */
system_prompt: string;
/** The list of tool definitions this Session exposes to the model (full schema, matching what's sent to the LLM). */
tools: ToolDefinition[];
/** The model's thinking level (from system_config.model.thinking_level; "default" when unconfigured). */
thinking_level: string;
/** Absolute path to the Agent State. */
agent_state: string;
/** Absolute path to the Workspace. */
workspace: string;
}
// ---------------------------------------------------------------------------
// model_msg — complete messages
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "model_msg: complete payloads"
export interface TextPayload {
type: "text";
role: Role;
text: string;
stop_reason?: StopReason;
/** Provider fidelity field: text phase marker (e.g. GPT-5 segments by phase), kept as-is and restored verbatim. */
phase?: string | null;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ImageUrlPayload {
type: "image_url";
role: "user";
/** A web URL or a base64 data URL. */
image_url: string;
stop_reason?: StopReason;
}
export interface InlineDataPayload {
type: "inline_data";
role: Role;
/** Base64-encoded bytes. */
data: string;
mime_type: string;
stop_reason?: StopReason;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ThinkingPayload {
type: "thinking";
role: "assistant";
thinking: string;
stop_reason?: StopReason;
/**
* Provider fidelity field: thinking-block signature (Claude thinking blocks / redacted
* thinking, GPT-5 encrypted reasoning, etc. — **required** when some models replay history),
* kept as-is and restored verbatim — losing it breaks Session resumption.
*/
signature?: string;
}
export interface InlineThinkingPayload {
type: "inline_thinking";
role: "assistant";
/** Base64-encoded bytes. */
data: string;
mime_type: string;
stop_reason?: StopReason;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ToolCallPayload {
type: "tool_call";
role: "assistant";
name: string;
/** Tool arguments as a JSON string. */
arguments: string;
tool_call_id: string;
stop_reason?: StopReason;
/** Provider fidelity field: signature, kept as-is and restored verbatim. */
signature?: string;
}
export interface ToolCallOutputPayload {
type: "tool_call_output";
role: "user";
output: string;
/**
* Images carried by the tool output (optional): each is a `data:<mime>;base64,...` data URL,
* fed back to the model alongside the text (e.g. images read by read_image). Images aren't
* incremental: the streaming path carries the whole set once via a single delta (see
* `PartialToolCallOutputPayload.images`), and the complete message carries them again — the
* streamed-and-joined result equals the complete message.
*/
images?: string[];
tool_call_id: string;
stop_reason?: StopReason;
}
// ---------------------------------------------------------------------------
// model_msg — streaming partial_* messages
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "model_msg: streaming partials"
export interface PartialTextPayload {
type: "partial_text";
role: "assistant";
event_type: StreamEventType;
text: string;
stop_reason?: StopReason;
}
export interface PartialThinkingPayload {
type: "partial_thinking";
role: "assistant";
event_type: StreamEventType;
thinking: string;
stop_reason?: StopReason;
}
export interface PartialToolCallPayload {
type: "partial_tool_call";
role: "assistant";
event_type: StreamEventType;
name: string;
/** Incremental fragment of the arguments JSON. */
arguments: string;
tool_call_id: string;
stop_reason?: StopReason;
}
export interface PartialToolCallOutputPayload {
type: "partial_tool_call_output";
role: "user";
event_type: StreamEventType;
output: string;
/** Images carried by the tool output (optional): images aren't incremental, carried as a whole by a single delta (consistent with the complete message). */
images?: string[];
tool_call_id: string;
stop_reason?: StopReason;
}
// ---------------------------------------------------------------------------
// event_msg
// ---------------------------------------------------------------------------
// Docs: /docs/omni-message § "event_msg"
export interface ApprovalDecisionPayload {
type: "approval_decision";
decision: ApprovalDecision;
tool_call_id: string;
}
export interface AbortPayload {
type: "abort";
reason?: string | null;
}
export interface TokenUsagePayload {
type: "token_usage";
/** Current Session cumulative token usage. */
session: TokenCounts;
/** Token usage for the most recent Request. */
request: TokenCounts;
}
/**
* Request boundary event: the boundary of one LLM Request, produced **in pairs** by
* `context_engine` and written to Trace. `request_end`
* with `status` of `completed` means the turn has been committed by AgentHub — this is the
* mechanical criterion Trace replay (Session resumption) uses to determine whether a turn was
* committed, and it also gives performance analysis a basis for Request latency and turn counts.
* A compaction request produces this same event pair too (written to Trace only, not streamed).
*/
export interface RequestBeginPayload {
type: "request_begin";
}
export interface RequestEndPayload {
type: "request_end";
/** Terminal state of this Request (reuses the five StopReason values, sharing its source with this turn's complete message's stop_reason / LLMOutcome). */
status: StopReason;
}
/** Compaction trigger reason: context threshold / turn-count threshold / user-initiated request. */
export type CompactionReason = "context" | "turns" | "manual";
/** Context compaction mode: summary relay / direct discard. */
export type CompactionMode = "summarize" | "discard";
/**
* Compaction boundary event: the compaction process exposes
* only this event pair to Human, produced **in pairs** by `context_engine`. Both `reason` and
* `mode` are carried on both events, for stateless frontend rendering; `status` reuses the
* five-value `StopReason` protocol (compaction converges to a terminal state, taking
* `completed` / `failed` / `aborted` in practice — `timeout` / `malformed` are handled internally
* by the compaction request's existing retry mechanism, collapsing to `failed` once retries are
* exhausted).
*/
export interface CompactionBeginPayload {
type: "compaction_begin";
reason: CompactionReason;
mode: CompactionMode;
/** Current context token usage (the most recent token_usage's request.total). */
context: number;
/** Session cumulative turn count. */
turns: number;
}
export interface CompactionEndPayload {
type: "compaction_end";
reason: CompactionReason;
mode: CompactionMode;
/** Compaction result; non-`completed` means compaction was abandoned and the original context was kept. */
status: StopReason;
}
/**
* Subagent pointer event: when the parent Session spawns a
* **direct** child session, `context_engine` writes this to the parent Trace (not streamed),
* recording only the child session's Session id — the child session's other details live in its
* own Trace's `session_meta`. When the session is reopened, the server uses this to recursively
* expand the child Trace and reconstruct the `origin` chain; a grandchild session's pointer is
* recorded by the child Trace itself.
*/
export interface SubagentPayload {
type: "subagent";
/** The direct child session's Session id. */
session_id: string;
}
// ---------------------------------------------------------------------------
// Union types and the message envelope
// ---------------------------------------------------------------------------
/** Complete model_msg payload (written to Trace and exposed externally). */
export type CompleteModelPayload =
| TextPayload
| ImageUrlPayload
| InlineDataPayload
| ThinkingPayload
| InlineThinkingPayload
| ToolCallPayload
| ToolCallOutputPayload;
/** Streaming model_msg payload. */
export type PartialModelPayload =
| PartialTextPayload
| PartialThinkingPayload
| PartialToolCallPayload
| PartialToolCallOutputPayload;
export type ModelPayload = CompleteModelPayload | PartialModelPayload;
export type EventPayload =
| ApprovalDecisionPayload
| AbortPayload
| RequestBeginPayload
| RequestEndPayload
| TokenUsagePayload
| CompactionBeginPayload
| CompactionEndPayload
| SubagentPayload;
export type OmniPayload = SessionMetaPayload | ModelPayload | EventPayload;
/** The unified message envelope. */
export interface OmniMessage<P extends OmniPayload = OmniPayload> {
/** ISO 8601 UTC timestamp. */
timestamp: string;
type: OmniMessageType;
payload: P;
/** Nested-origin marker: the chain of child Session ids ordered outer-to-inner; absent = from the main Session (see MessageOrigin). */
origin?: MessageOrigin[];
}
// Convenience aliases for concrete message types --------------------------------
export type SessionMetaMessage = OmniMessage<SessionMetaPayload>;
export type ModelMessage = OmniMessage<ModelPayload>;
export type EventMessage = OmniMessage<EventPayload>;
export type CompleteModelMessage = OmniMessage<CompleteModelPayload>;
export type PartialModelMessage = OmniMessage<PartialModelPayload>;
// ---------------------------------------------------------------------------
// Runtime discrimination helpers
// ---------------------------------------------------------------------------
/** The set of type values for streaming partial_* payloads. */
const PARTIAL_PAYLOAD_TYPES = [
"partial_text",
"partial_thinking",
"partial_tool_call",
"partial_tool_call_output",
] as const;
export function isPartialPayload(p: OmniPayload): p is PartialModelPayload {
return (PARTIAL_PAYLOAD_TYPES as readonly string[]).includes((p as { type?: string }).type ?? "");
}
export function isModelMessage(msg: OmniMessage): msg is ModelMessage {
return msg.type === "model_msg";
}
export function isEventMessage(msg: OmniMessage): msg is EventMessage {
return msg.type === "event_msg";
}
export function isSessionMeta(msg: OmniMessage): msg is SessionMetaMessage {
return msg.type === "session_meta";
}
/** A complete model_msg (not partial_*), i.e. a message that can be written to Trace. */
export function isCompleteModelMessage(msg: OmniMessage): msg is CompleteModelMessage {
return msg.type === "model_msg" && !isPartialPayload(msg.payload);
}
+124
View File
@@ -0,0 +1,124 @@
/**
* Session title generation: an **out-of-band, one-off request** that generates
* a short title from the first-turn conversation text.
*
* Called by `session.generateTitle()`: sends one request using the bare LLM for the session's
* Model (no tools, no system prompt, thinking off), without writing history or Trace. Material
* defaults to what the Session self-captures during run (see session.ts); this module is only
* responsible for the prompt format, driving the one-off request, and sanitizing the result —
* when to generate a title and where to store it is decided by the host (Web server / CLI).
*/
import { userText } from "./omnimessage/index.js";
import type {
OmniMessage,
TextPayload,
TokenCounts,
TokenUsagePayload,
} from "./omnimessage/index.js";
import type { LLMInterface } from "./interfaces.js";
/** Cap on conversation text spliced into the title request (user/model each truncated separately, to control cost). */
const EXCERPT_MAX_CHARS = 2000;
/** Cap on title length (fallback truncation for when the model occasionally ignores the constraint). */
const TITLE_MAX_CHARS = 30;
export interface SessionTitleResult {
/** The sanitized title; null when material is insufficient, the request fails, or the output is empty. */
title: string | null;
/** Token consumption for this request (accumulated token_usage.request); null if no request occurred or there's no usage. */
usage: TokenCounts | null;
}
/**
* Assembles the title-generation Prompt (exported for host/test assertion use). Uses English
* instructions to avoid polluting the title's language, and requires the **title to be in the
* same language as the conversation** (English conversation gets an English title, Chinese
* conversation gets a Chinese title); when assistant material is empty, it relies on the user
* request alone.
*/
export function buildTitlePrompt(userExcerpt: string, assistantExcerpt: string): string {
const clip = (s: string) => (s.length > EXCERPT_MAX_CHARS ? s.slice(0, EXCERPT_MAX_CHARS) : s);
const lines = [
"Generate a concise title for the conversation below.",
"Rules:",
"- Write the title in the SAME language the user is using.",
"- Keep it short: at most 6 words, or ~16 characters for CJK.",
"- Output ONLY the title text — no quotes, no trailing punctuation, no explanation.",
"",
"[User]",
clip(userExcerpt),
];
if (assistantExcerpt.trim()) {
lines.push("", "[Assistant]", clip(assistantExcerpt));
}
return lines.join("\n");
}
/** Sanitizes model output into a title: strips leading/trailing quotes/brackets and trailing punctuation (until stable), collapses whitespace, and truncates if too long; returns null for an empty result. */
export function sanitizeTitle(raw: string): string | null {
let t = raw.replace(/\s+/g, " ").trim();
// Stripping quotes can expose more punctuation underneath (or vice versa), so strip repeatedly until stable.
for (let prev = ""; prev !== t;) {
prev = t;
t = t
.replace(/^["'“”‘’「」『』《》〈〉【】()()\s]+/, "")
.replace(/["'“”‘’「」『』《》〈〉【】()()\s]+$/, "")
.replace(/[。..!!??;;,,、::]+$/, "")
.trim();
}
if (!t) return null;
return t.length > TITLE_MAX_CHARS ? t.slice(0, TITLE_MAX_CHARS) : t;
}
/**
* Drives a single title-generation request: collects model text and token_usage, and resolves
* based on the outcome. Generation only requires user material (assistant material may be
* empty — a pure tool-only turn can still get a title); no request is sent if user material is
* empty; `title` is null if the request doesn't complete (any usage already produced is still
* returned).
*/
export async function generateTitleWithLLM(
llm: LLMInterface,
args: { userText: string; assistantText: string; signal?: AbortSignal },
): Promise<SessionTitleResult> {
if (!args.userText.trim()) {
return { title: null, usage: null };
}
const prompt = buildTitlePrompt(args.userText, args.assistantText);
const gen = llm.streamGenerate({
newMessages: [userText(prompt)],
...(args.signal ? { signal: args.signal } : {}),
});
let collected = "";
let usage: TokenCounts | null = null;
for (;;) {
const step = await gen.next();
if (step.done) {
if (step.value.status !== "completed") return { title: null, usage };
break;
}
const msg = step.value;
if (isAssistantText(msg)) collected += msg.payload.text;
if (isTokenUsage(msg)) {
const r = msg.payload.request;
usage = usage
? {
cache_read: usage.cache_read + r.cache_read,
cache_write: usage.cache_write + r.cache_write,
output: usage.output + r.output,
total: usage.total + r.total,
}
: { ...r };
}
}
return { title: sanitizeTitle(collected), usage };
}
function isAssistantText(msg: OmniMessage): msg is OmniMessage<TextPayload> {
const payload = msg.payload as { type?: string; role?: string };
return msg.type === "model_msg" && payload.type === "text" && payload.role === "assistant";
}
function isTokenUsage(msg: OmniMessage): msg is OmniMessage<TokenUsagePayload> {
return msg.type === "event_msg" && (msg.payload as { type?: string }).type === "token_usage";
}
+250
View File
@@ -0,0 +1,250 @@
/**
* Session — a continuous conversation context under the same Agent and Workspace.
*
* Human is the SDK's input/output boundary: there is no "Human
* implementation/interface".
* - Input: the OmniMessage list (Prompt) passed to `run(newMessages, opts?)`, plus the abort
* signal `signal` and the per-call approval callback `approve` in `opts`;
* - Output: `run` streams OmniMessage via an async generator.
*
* Approval is a **within-turn interaction**: as soon as a tool_call finishes streaming, `approve`
* is requested immediately, and it executes if allowed. Approvals for multiple tools happen one
* at a time, but execution doesn't block the generation/approval of subsequent tools (execution
* can overlap). GenerativeModel maintains history across turns/Tasks. A Task ends when a turn no
* longer produces a tool_call (final reply).
*
* Rendering tool calls is not Session/core's responsibility: the CLI / Web frontend renders it
* from the streamed OmniMessage on its own.
* Docs: /docs/agent-loop; /docs/interfaces § "The Human boundary".
*/
import { sessionMeta } from "./omnimessage/index.js";
import type { OmniMessage, SessionMetaPayload, TokenCounts } from "./omnimessage/index.js";
import { imagesToScratchpadPaths } from "./internal/session-support.js";
import type { EnvironmentInterface, LLMInterface, ToolPermission } from "./interfaces.js";
import { generateTitleWithLLM } from "./session-title.js";
import type { SessionTitleResult } from "./session-title.js";
import { ContextEngine } from "./engine/context-engine.js";
import type {
CompactAvailability,
CompactionSettings,
EngineInitialState,
RunOptions,
TraceSink,
} from "./engine/context-engine.js";
export interface SessionConfig {
/** Session metadata (session_id / provider / model_id / model_context_window / system_prompt / tools / thinking_level / agent_state / workspace). */
meta: SessionMetaPayload;
llm: LLMInterface;
environment: EnvironmentInterface;
trace?: TraceSink;
maxTurns?: number;
/** Creates a new LLM object after compaction (carries over the Session's accumulated Token count); context compaction is unavailable if not provided. */
createLLM?: (sessionTokens: TokenCounts) => LLMInterface;
/**
* Factory for the bare LLM used by out-of-band, one-off requests (same Model/credential as
* the session; no tools, no system prompt, thinking off): used for meta-requests such as
* `generateTitle`; if not provided, `generateTitle` returns null.
*/
createBareLLM?: () => LLMInterface;
/** Context compaction settings (defaults are filled in by the composition layer); only takes effect when provided together with `createLLM`. */
compaction?: CompactionSettings;
/** Session resume: `session_meta` is already in the original Trace file, so it isn't written again on the first run (avoids duplication). */
metaAlreadyWritten?: boolean;
/** Session resume: the engine's initial state derived from Trace replay (carry-over / accumulated stats, etc.). */
initialEngineState?: EngineInitialState;
/** Session resume: the full historical messages of the current context (for rendering, including interrupted turns and their markers), for frontend display. */
resumedHistory?: OmniMessage[];
/**
* Set when the session's model doesn't support images (the composition layer decides this via
* ModelEntry.vision): images in `run` input are saved to this directory (the session's
* scratchpad), and the path is appended to the user text instead — the model views the image
* via describe_image, and images never enter the session history directly (some providers
* return a 400 outright on image input).
*/
inputImagesDir?: string;
}
/** Cap on captured title material (chars per side, matching buildTitlePrompt's truncation); stops accumulating once exceeded. */
const TITLE_MATERIAL_LIMIT = 2000;
/**
* Accumulates title material: the body text of complete text messages from the main session
* (no origin) — thinking and tool calls naturally don't count — and stops once the cap is hit.
*/
function appendTitleText(base: string, msg: OmniMessage, role: "user" | "assistant"): string {
if (base.length >= TITLE_MATERIAL_LIMIT) return base;
if (msg.origin && msg.origin.length > 0) return base;
const p = msg.payload as { type?: string; role?: string; text?: string };
if (msg.type !== "model_msg" || p.type !== "text" || p.role !== role || !p.text) return base;
return base ? `${base}\n${p.text}` : p.text;
}
export class Session {
readonly sessionId: string;
/** The session model's provider group (paired with `modelId` to form the model reference). */
readonly provider: string;
/** The session model's upstream model_id (the request id sent to AgentHub). */
readonly modelId: string;
readonly workspaceDir: string;
/** Session resume: the full historical messages of the current context (for rendering); undefined for a non-resumed Session. */
readonly resumedHistory?: OmniMessage[];
private readonly engine: ContextEngine;
private readonly environment: EnvironmentInterface;
private readonly trace?: TraceSink;
private readonly meta: OmniMessage;
private readonly createBareLLM?: () => LLMInterface;
private readonly inputImagesDir?: string;
private metaWritten = false;
/** Title material (used by `generateTitle` as the default): the user input and model body text of the first Task that contains user text. */
private titleUserText = "";
private titleAssistantText = "";
/** Material-frozen flag: becomes true once the first Task containing user text finishes; subsequent runs stop accumulating. */
private titleMaterialFrozen = false;
constructor(config: SessionConfig) {
this.sessionId = config.meta.session_id;
this.provider = config.meta.provider;
this.modelId = config.meta.model_id;
this.workspaceDir = config.meta.workspace;
this.environment = config.environment;
this.trace = config.trace;
this.meta = sessionMeta(config.meta);
this.metaWritten = config.metaAlreadyWritten ?? false;
if (config.resumedHistory) this.resumedHistory = config.resumedHistory;
if (config.createBareLLM) this.createBareLLM = config.createBareLLM;
if (config.inputImagesDir) this.inputImagesDir = config.inputImagesDir;
this.engine = new ContextEngine({
llm: config.llm,
environment: config.environment,
...(config.trace ? { trace: config.trace } : {}),
...(config.maxTurns !== undefined ? { maxTurns: config.maxTurns } : {}),
// Context compaction: new LLM factory + resolved settings + writes session_meta at the start of the new Trace file after splitting.
...(config.createLLM ? { createLLM: config.createLLM } : {}),
...(config.compaction ? { compaction: config.compaction } : {}),
...(config.initialEngineState ? { initialState: config.initialEngineState } : {}),
sessionMeta: this.meta,
});
}
/**
* Runs a Task to completion and streams out OmniMessage. `newMessages` is this call's Prompt
* (only the newly added input); `opts` carries the abort signal `signal` and the per-call
* approval callback `approve` (the engine calls it once per tool_call within a turn).
* On the first run, `session_meta` is written to the Trace first.
*
* A single `run` automatically drives the whole ReAct loop: consuming the LLM stream,
* approving and executing tools one at a time, feeding results back for the next turn,
* until a turn no longer produces a tool_call (Task ends) or it's aborted.
* Docs: /docs/agent-loop § "The loop at a glance".
*/
async *run(newMessages: OmniMessage[], opts?: RunOptions): AsyncGenerator<OmniMessage> {
// Model doesn't support images: input images are saved to disk first (session scratchpad),
// then the path is appended to the text before it reaches the engine/Trace.
if (this.inputImagesDir) {
newMessages = await imagesToScratchpadPaths(newMessages, this.inputImagesDir);
}
await this.ensureMetaWritten();
// Self-captures title material (the title is derived from the first-turn
// conversation text): while material isn't frozen yet, collect this call's user text and
// the produced model text; freezes once the first Task containing user text finishes, so
// the title reflects the start of the conversation.
const capture = !this.titleMaterialFrozen;
if (capture) {
for (const m of newMessages) {
this.titleUserText = appendTitleText(this.titleUserText, m, "user");
}
}
for await (const msg of this.engine.run(newMessages, opts)) {
if (capture) {
this.titleAssistantText = appendTitleText(this.titleAssistantText, msg, "assistant");
}
yield msg;
}
if (capture && this.titleUserText.trim()) this.titleMaterialFrozen = true;
}
/**
* User-initiated request to compact context (e.g. a CLI command): reuses the automatic
* compaction flow but skips the threshold check (reason=manual). Only callable at Task
* boundaries (between runs); streams out paired `compaction` events. The summarize digest
* becomes the prefix of the next `run`'s input (merged with the next user Prompt). A no-op
* if compaction isn't configured.
* Docs: /docs/agent-loop § "Compaction".
*/
async *compact(opts?: { signal?: AbortSignal }): AsyncGenerator<OmniMessage> {
yield* this.engine.compact(opts);
}
/**
* Whether compaction is possible, and why not if not (see ContextEngine.compactability).
* When the result isn't `ok`, `compact()` is a no-op and yields no messages — callers should
* give feedback based on this rather than triggering a silent, fruitless compaction.
*/
compactability(): CompactAvailability {
return this.engine.compactability();
}
/** Writes `session_meta` to the Trace before the first run/compaction; best-effort — failure doesn't interrupt the run. */
private async ensureMetaWritten(): Promise<void> {
if (this.metaWritten) return;
if (this.trace) {
try {
await this.trace.write(this.meta);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
process.stderr.write(`[trace] session_meta write failed: ${message}\n`);
}
}
this.metaWritten = true;
}
/**
* Out-of-band, one-off request that generates a short title from the first-turn conversation
* text: sends one request using the bare LLM for the session's Model (no
* tools, no system prompt, thinking off), **without writing history or Trace**. Material
* defaults to the first Task text self-captured by the Session (user input and model body
* text collected during run; thinking and tool calls don't count), so callers don't need to
* supply it; `material` can override this (e.g. when a host generates a title for a
* sub-session — the material is that sub-session's own conversation). `title` is null if the
* material is empty, the request fails, or the composition layer didn't supply a bare LLM
* factory. Token consumption is returned via `usage` for the host to account for.
* Docs: /docs/agent-loop § "Side channels".
*/
async generateTitle(args?: {
/** Material override; defaults to the Session's self-captured material. */
material?: { userText: string; assistantText: string };
signal?: AbortSignal;
}): Promise<SessionTitleResult> {
if (!this.createBareLLM) return { title: null, usage: null };
const material = args?.material ?? {
userText: this.titleUserText,
assistantText: this.titleAssistantText,
};
return generateTitleWithLLM(this.createBareLLM(), {
...material,
...(args?.signal ? { signal: args.signal } : {}),
});
}
/** Queries a tool's permission level (for the frontend to determine permission mode); returns undefined for unknown tools. */
toolPermission(name: string): ToolPermission | undefined {
return this.environment.toolPermission(name);
}
/** This Session's session_meta message (used e.g. by host tools to forward nested-session metadata to a parent session). */
get metaMessage(): OmniMessage {
return this.meta;
}
/**
* Releases runtime resources held by the Session: kills long-running command sessions
* managed by the Environment. The host calls this when the Session ends (CLI exit, Web
* session close) to avoid leaking background processes into the host process's lifetime.
* Optional, idempotent.
*/
dispose(): void {
this.environment.dispose?.();
}
}
+418
View File
@@ -0,0 +1,418 @@
/**
* Loading and initialization of Agent State (semantics modeled on Hugging Face model loading).
*
* - Initializes when the target Agent directory is empty (no `system_config.yaml`): creates
* `agent_state/`, `tools/`, `memory/`, `skills/`, and the sibling `scratchpad/`, and writes
* the default `system_config.yaml` and `AGENTS.md`.
* - Otherwise loads the existing system config and editable Prompt for the given `agentId`.
*
* The full runtime Prompt is rendered from the system-level Prompt template in
* `system_config.yaml`; placeholders in the template are replaced with `AGENTS.md` and the
* concrete Session runtime environment fields. Built-in tools and MCP Server config
* come from `system_config.yaml`.
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseYaml, stringify as stringifyYaml } from "yaml";
import {
loadLibrarySkills,
parseSkillFrontmatter,
type SkillMetadata,
} from "@prismshadow/penguin-skills";
import type { ToolConfig, ToolDefinitionConfig } from "../interfaces.js";
import {
AGENT_ID_PLACEHOLDER,
AGENTS_MD_PLACEHOLDER,
VAULT_KEYS_PLACEHOLDER,
SKILL_METADATA_PLACEHOLDER,
CWD_PLACEHOLDER,
DATE_PLACEHOLDER,
defaultAgentsMd,
defaultSystemConfig,
OS_VERSION_PLACEHOLDER,
PLATFORM_PLACEHOLDER,
PROJECT_DIR_PLACEHOLDER,
SESSION_ID_PLACEHOLDER,
type SystemConfig,
} from "./default-config.js";
import { builtinProjectAgentPresets, type AgentPreset } from "./builtin-agents.js";
import { provisionExampleBenchmark } from "./example-benchmark.js";
import {
agentsMdPath,
agentStateDir,
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
memoryDir,
resolveRoot,
scratchpadDir,
skillsDir,
systemConfigPath,
toolsDir,
} from "./paths.js";
/** project_id / agent_id / skill_name only allow letters, digits, underscore `_`, and hyphen `-` (prevents path traversal). */
const ID_PATTERN = /^[A-Za-z0-9_-]+$/;
export type IdKind = "project_id" | "agent_id" | "skill_name";
export function isValidId(id: string): boolean {
return ID_PATTERN.test(id);
}
export function assertValidId(kind: IdKind, id: string): void {
if (!ID_PATTERN.test(id)) {
throw new Error(
`Invalid ${kind} ${JSON.stringify(id)}: only letters, digits, "_" and "-" are allowed.`,
);
}
}
/** A loaded Agent State handle. */
export interface AgentState {
root: string;
projectId: string;
agentId: string;
stateDir: string;
systemConfig: SystemConfig;
agentsMd: string;
}
export interface SessionEnvironmentValues {
sessionId: string;
cwd: string;
/** The Agent id this Session belongs to (system Prompt placeholder {{AGENT_ID}}). */
agentId: string;
/** Absolute path to this Project's directory (system Prompt placeholder {{PROJECT_DIR}}; Agent State/scratchpad paths are derived from it). */
projectDir: string;
platform: string;
osVersion: string;
date: string;
}
/**
* Loads or initializes Agent State.
*
* When root/project/agent are omitted, `resolveRoot()` and the default constants are used. If
* `system_config.yaml` doesn't exist, the directory is treated as empty and initialized;
* otherwise the existing content is loaded. `preset` only takes effect on the initialization
* path (name/description/AGENTS.md overrides and extra Skills) and is ignored when loading an
* existing Agent — existing config is never overwritten.
*/
export async function loadOrInitAgentState(opts?: {
agentId?: string;
projectId?: string;
root?: string;
preset?: AgentPreset;
}): Promise<AgentState> {
const root = opts?.root ?? resolveRoot();
const projectId = opts?.projectId ?? DEFAULT_PROJECT_ID;
const agentId = opts?.agentId ?? DEFAULT_AGENT_ID;
// Validate before building paths, to prevent path traversal.
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
const stateDir = agentStateDir(root, projectId, agentId);
const configPath = systemConfigPath(root, projectId, agentId);
const mdPath = agentsMdPath(root, projectId, agentId);
let systemConfig: SystemConfig;
let agentsMd: string;
if (await fileExists(configPath)) {
// Load path: read the existing system_config.yaml and AGENTS.md.
const rawConfig = await fs.readFile(configPath, "utf8");
const parsed = parseYaml(rawConfig) as unknown;
// Defensive check: if the file is empty/corrupted, parseYaml may return null/a non-object,
// or system_prompt may be missing — otherwise "undefined" would get spliced into the system
// Prompt. Throw a clear error when validation fails.
if (
parsed === null ||
typeof parsed !== "object" ||
typeof (parsed as SystemConfig).system_prompt !== "string"
) {
throw new Error(`Agent State 配置非法:${configPath} 为空、损坏或缺少 system_prompt 字段。`);
}
systemConfig = parsed as SystemConfig;
agentsMd = (await fileExists(mdPath)) ? await fs.readFile(mdPath, "utf8") : defaultAgentsMd();
} else {
// Init path: create the directory structure and write default config (preset only takes effect here).
await Promise.all([
fs.mkdir(stateDir, { recursive: true }),
fs.mkdir(toolsDir(root, projectId, agentId), { recursive: true }),
fs.mkdir(memoryDir(root, projectId, agentId), { recursive: true }),
fs.mkdir(skillsDir(root, projectId, agentId), { recursive: true }),
fs.mkdir(scratchpadDir(root, projectId, agentId), { recursive: true }),
]);
const preset = opts?.preset;
systemConfig = {
...defaultSystemConfig(),
...(preset?.name !== undefined ? { name: preset.name } : {}),
...(preset?.description !== undefined ? { description: preset.description } : {}),
};
agentsMd = preset?.agentsMd ?? defaultAgentsMd();
// Only installs the Skills specified by preset (a plain newly created Agent gets none
// pre-installed). A default_agent with no
// preset (e.g. created on first CLI run) still gets every Skill in the library pre-installed
// — the install policy follows Agent identity, not whether creation came from the server or
// was done directly via SDK/CLI.
// Skills have no dedicated tool: metadata is injected via {{SKILL_METADATA}}, and the model
// reads SKILL.md with shell and follows it.
const skills =
opts?.preset === undefined && agentId === DEFAULT_AGENT_ID
? loadLibrarySkills()
: (opts?.preset?.skills ?? []);
await Promise.all([
fs.writeFile(mdPath, agentsMd, "utf8"),
...skills.map((skill) => installSkill(root, projectId, agentId, skill)),
// The example Benchmark is only provisioned alongside default_agent (so the evaluation
// center has data out of the box): idempotently skipped if benchmarks/ already exists,
// and not created for plain Agents.
...(agentId === DEFAULT_AGENT_ID
? [provisionExampleBenchmark(root, projectId, agentId)]
: []),
]);
// system_config.yaml is written last: its existence is the "initialization complete" marker
// (the load/init decision point). If this fails partway (disk full / crash), the next run
// still takes the init path and self-heals, so no half-initialized state with missing Skills is left behind.
await fs.writeFile(configPath, stringifyYaml(systemConfig), "utf8");
}
return { root, projectId, agentId, stateDir, systemConfig, agentsMd };
}
/**
* Initializes a Project's built-in Agent (the only built-in Agent: default_agent).
*
* Calls loadOrInitAgentState for each one: an Agent whose directory already exists (including a
* default_agent created earlier by the CLI) is only loaded, never overwritten (preset only
* takes effect on initialization). Returns the list of built-in Agent ids.
*/
export async function provisionProjectAgents(opts?: {
root?: string;
projectId?: string;
}): Promise<string[]> {
const agentIds: string[] = [];
for (const { agentId, preset } of builtinProjectAgentPresets()) {
await loadOrInitAgentState({
...(opts?.root !== undefined ? { root: opts.root } : {}),
...(opts?.projectId !== undefined ? { projectId: opts.projectId } : {}),
agentId,
preset,
});
agentIds.push(agentId);
}
return agentIds;
}
/**
* The vault key-name list: the replacement value for `{{VAULT_KEYS}}`, one `- KEY` per line;
* returns an empty string when there are no keys.
* **Contains only key names, never values** — values are only injected into the exec_command
* subprocess environment, never the model context. The statement of the vault's purpose is part
* of the default template body (the # Vault section) and is kept even with no vault.
*/
function vaultKeysList(keys: string[]): string {
return keys.map((key) => `- ${key}`).join("\n");
}
/**
* Installs a Skill into the target Agent: writes `skills/<name>/SKILL.md` verbatim (the full
* SKILL.md content including frontmatter, ensuring a trailing newline); if the directory
* already exists, it's overwritten (reinstalling = updating to the latest content). An optional
* icon.svg is written alongside SKILL.md; if this install doesn't
* include an icon, any old icon.svg is removed, preserving "overwrite update" semantics (the
* directory content matches the Skill being installed).
* Docs: /docs/skills § "Installation and storage".
*/
export async function installSkill(
root: string,
projectId: string,
agentId: string,
skill: { name: string; content: string; icon?: string },
): Promise<void> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
assertValidId("skill_name", skill.name);
const dir = path.join(skillsDir(root, projectId, agentId), skill.name);
await fs.mkdir(dir, { recursive: true });
const content = skill.content.endsWith("\n") ? skill.content : `${skill.content}\n`;
const iconPath = path.join(dir, "icon.svg");
await Promise.all([
fs.writeFile(path.join(dir, "SKILL.md"), content, "utf8"),
skill.icon !== undefined
? fs.writeFile(iconPath, skill.icon, "utf8")
: fs.rm(iconPath, { force: true }),
]);
}
/** Uninstalls a Skill: deletes the entire `skills/<name>/` directory; idempotent, no error if it doesn't exist. */
export async function removeSkill(
root: string,
projectId: string,
agentId: string,
name: string,
): Promise<void> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
assertValidId("skill_name", name);
await fs.rm(path.join(skillsDir(root, projectId, agentId), name), {
recursive: true,
force: true,
});
}
/** An installed Skill entry: frontmatter metadata (including an optional short description) + the optional icon.svg content in the directory. */
export interface InstalledSkill extends SkillMetadata {
/** The raw content of `skills/<name>/icon.svg` (a custom icon copied alongside SKILL.md at install time); the field is omitted when missing (the frontend falls back to a default book icon). */
icon?: string;
}
/**
* Lists the metadata of Skills installed on the target Agent: scans `skills/<name>/SKILL.md` and
* parses its frontmatter (optional fields like short_description(_zh) pass through as parsed),
* also reading the optional icon.svg content in the directory. Tolerant: a directory whose
* frontmatter fails to parse or is missing `name` falls back to
* `{ name: <directory name>, description: "", version: 1, updated: "" }`; a directory with no
* SKILL.md doesn't count as a Skill; returns [] if skills/ doesn't exist. Results are sorted by
* name (a stable order for both Prompt injection and API responses).
* Docs: /docs/skills § "Installation and storage".
*/
export async function listInstalledSkills(
root: string,
projectId: string,
agentId: string,
): Promise<InstalledSkill[]> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
const dir = skillsDir(root, projectId, agentId);
let entries;
try {
entries = await fs.readdir(dir, { withFileTypes: true });
} catch {
return [];
}
const skills: InstalledSkill[] = [];
for (const entry of entries) {
if (!entry.isDirectory()) continue;
let raw: string;
try {
raw = await fs.readFile(path.join(dir, entry.name, "SKILL.md"), "utf8");
} catch {
continue;
}
let icon: string | undefined;
try {
icon = await fs.readFile(path.join(dir, entry.name, "icon.svg"), "utf8");
} catch {
// icon.svg is optional: missing means no custom icon.
}
// The directory name is the Skill's identity (install / uninstall / Prompt read guidance all
// address by directory name): frontmatter only supplies display fields like description; when
// its `name` doesn't match the directory name (a hand-written or network-sourced Skill), the
// directory name always wins — otherwise the model would read a nonexistent path using the
// injected name, and the API couldn't uninstall it either.
const parsed = parseSkillFrontmatter(raw);
skills.push({
...(parsed ?? { description: "", version: 1, updated: "" }),
name: entry.name,
...(icon !== undefined ? { icon } : {}),
});
}
return skills.sort((a, b) => a.name.localeCompare(b.name));
}
/**
* Skill metadata section: the replacement value for `{{SKILL_METADATA}}`, one line per Skill in
* the form `- \`name\` — description` (just the name when description is empty); an empty array
* returns an empty string. The full body is read by the model on demand via shell.
*/
export function skillMetadataSection(skills: SkillMetadata[]): string {
return skills
.map((s) => (s.description ? `- \`${s.name}\` — ${s.description}` : `- \`${s.name}\``))
.join("\n");
}
/**
* Renders the complete runtime system Prompt: substitutes `AGENTS.md`, vault key names, Skill
* metadata, and the concrete Session runtime environment placeholders into the system Prompt
* template. The assembly layer only does placeholder substitution and adds no extra text —
* wrapper text such as `<developer_instructions>` and the # Vault / # Skills statements are
* written directly into the system Prompt template itself (the Prompt is fully
* transparent and editable via `system_config.yaml`). Other files in Agent State / Workspace are
* never auto-injected.
*
* `{{VAULT_KEYS}}` is replaced with the vault key-name list (an empty string if empty/not
* provided): this lets the model know which APIs requiring a key it can call; values are never
* injected. `{{SKILL_METADATA}}` is replaced with the installed Skills' metadata lines (an empty
* string if empty/not provided). A custom template that removes a placeholder gets no
* corresponding content injected.
* Docs: /docs/configuration § "System prompt placeholders".
*/
export function assembleSystemPrompt(
state: AgentState,
sessionEnvironment?: SessionEnvironmentValues,
vaultKeys?: string[],
skillMetadata?: SkillMetadata[],
): string {
return state.systemConfig.system_prompt
.split(AGENTS_MD_PLACEHOLDER)
.join(state.agentsMd.trim())
.split(VAULT_KEYS_PLACEHOLDER)
.join(vaultKeysList(vaultKeys ?? []))
.split(SKILL_METADATA_PLACEHOLDER)
.join(skillMetadataSection(skillMetadata ?? []))
.split(AGENT_ID_PLACEHOLDER)
.join(sessionEnvironment?.agentId ?? state.agentId)
.split(PROJECT_DIR_PLACEHOLDER)
.join(sessionEnvironment?.projectDir ?? "")
.split(SESSION_ID_PLACEHOLDER)
.join(sessionEnvironment?.sessionId ?? "")
.split(CWD_PLACEHOLDER)
.join(sessionEnvironment?.cwd ?? "")
.split(PLATFORM_PLACEHOLDER)
.join(sessionEnvironment?.platform ?? "")
.split(OS_VERSION_PLACEHOLDER)
.join(sessionEnvironment?.osVersion ?? "")
.split(DATE_PLACEHOLDER)
.join(sessionEnvironment?.date ?? "")
.trim();
}
/**
* Builds the `ToolConfig` needed by Environment from Agent State.
*
* Both builtin tools and MCP Server config are taken from `system_config.yaml`; falls back to the
* default config when builtin tools are missing.
*/
/**
* Filters builtin tool entries by the session model's type: entries with `forModel: "vision"` are
* only used for models that support images (vision models), `forModel: "text-only"` is only for
* text-only models (e.g. choosing between read_image / describe_image); unlabeled entries are
* available to all models.
* Docs: /docs/tools § "Image tools".
*/
export function selectBuiltinToolsForModel(
tools: ToolDefinitionConfig[],
modelVision: boolean,
): ToolDefinitionConfig[] {
const kind = modelVision ? "vision" : "text-only";
return tools.filter((t) => t.forModel === undefined || t.forModel === kind);
}
export function buildToolConfig(state: AgentState): ToolConfig {
const systemTools = state.systemConfig.tools;
const builtin = systemTools?.builtin ?? defaultSystemConfig().tools?.builtin ?? [];
return {
customTools: builtin,
mcpServers: systemTools?.mcpServers ?? [],
};
}
async function fileExists(filePath: string): Promise<boolean> {
try {
await fs.access(filePath);
return true;
} catch {
return false;
}
}
+146
View File
@@ -0,0 +1,146 @@
/**
* Agent-level environment-variable vault (`<project>/agents/<agent_id>/agent_state/.vault.toml`).
*
* Key-value pairs such as third-party API keys, configured per Agent: injected into that Agent
* session's `exec_command` / `input_command` child-process environment, with key names disclosed
* to the model via the system Prompt while values never enter the model context. Carries the same
* trade-offs as a credential: stored in plaintext on disk, masked at the API layer. The file is
* created/removed together with the Agent directory; its absence is treated as an empty table;
* once emptied, the file is removed to avoid leaving a stray empty .vault.toml.
* Docs: /docs/configuration § "Vault".
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseToml, stringify as stringifyToml } from "smol-toml";
import { agentVaultPath } from "./paths.js";
import { assertValidId } from "./agent-state.js";
/** Vault key-name constraint: matches shell environment variable names (starts with a letter or underscore, followed by letters/digits/underscores only). */
const VAULT_KEY_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/;
/**
* Vault value length cap: since values are injected into the child-process environment, Linux
* caps a single env entry at roughly 128KB, and an oversized value would make every
* exec_command spawn for that Agent fail (E2BIG) — so it's rejected on the write side (core and
* the API layer share this same cap).
*/
export const VAULT_VALUE_MAX_LENGTH = 8192;
/** Whether a vault key name is valid (shell environment variable name rules). */
export function isValidVaultKey(key: string): boolean {
return VAULT_KEY_PATTERN.test(key);
}
/** Validates a vault key name, throwing if invalid (core and the API layer share this same rule). */
export function assertValidVaultKey(key: string): void {
if (!isValidVaultKey(key)) {
throw new Error(
`Invalid vault key ${JSON.stringify(key)}: only letters, digits and "_" are allowed, and it must not start with a digit.`,
);
}
}
/** Validates a vault value's length (see `VAULT_VALUE_MAX_LENGTH`), throwing if it exceeds the cap. */
export function assertValidVaultValue(key: string, value: string): void {
if (value.length > VAULT_VALUE_MAX_LENGTH) {
throw new Error(
`Vault value for ${key} is too long: ${value.length} > ${VAULT_VALUE_MAX_LENGTH} characters.`,
);
}
}
/**
* Reads the Agent vault: returns an empty table if the file doesn't exist.
* A hand-edited file is filtered by the same rule as the write side: only string values are
* accepted (numbers/dates etc. are ignored), and key names must follow shell variable name rules
* (invalid keys are always ignored — otherwise they'd get injected into the Prompt/child-process
* environment, and an invalid key surfaced by a GET view would make a full-table PUT 400, leaving
* the vault page unable to add or remove any further entries).
*/
export async function loadAgentVault(
root: string,
projectId: string,
agentId: string,
): Promise<Record<string, string>> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
let raw: string;
try {
raw = await fs.readFile(agentVaultPath(root, projectId, agentId), "utf8");
} catch {
return {};
}
const parsed: unknown = parseToml(raw) ?? {};
const vault: Record<string, string> = {};
if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) {
for (const [k, v] of Object.entries(parsed)) {
if (typeof v === "string" && isValidVaultKey(k)) vault[k] = v;
}
}
return vault;
}
/**
* Writes the full table to the Agent vault: validates all key names first; an empty table
* deletes the file (idempotent if it doesn't exist).
* The directory is created automatically if it doesn't exist (the vault can be configured even
* before the Agent is initialized).
*/
export async function saveAgentVault(
root: string,
projectId: string,
agentId: string,
vault: Record<string, string>,
): Promise<void> {
assertValidId("project_id", projectId);
assertValidId("agent_id", agentId);
for (const key of Object.keys(vault)) assertValidVaultKey(key);
const file = agentVaultPath(root, projectId, agentId);
if (Object.keys(vault).length === 0) {
await fs.rm(file, { force: true });
return;
}
await fs.mkdir(path.dirname(file), { recursive: true });
// The secret file is written to disk with mode 0600 (a hidden file blocks `ls`, not reads; mode
// only takes effect on creation, so chmod is applied to converge an existing file too).
await fs.writeFile(file, `${stringifyToml(vault)}\n`, { encoding: "utf8", mode: 0o600 });
await fs.chmod(file, 0o600);
}
/**
* Writes or updates one vault entry (added if it doesn't exist, overwritten if it does).
* The key name must follow shell environment variable name rules (see `isValidVaultKey`), and the
* value length is constrained by `VAULT_VALUE_MAX_LENGTH`; throws if invalid. Returns the updated
* vault.
*/
export async function setVaultEntry(
root: string,
projectId: string,
agentId: string,
key: string,
value: string,
): Promise<Record<string, string>> {
assertValidVaultKey(key);
assertValidVaultValue(key, value);
const vault = await loadAgentVault(root, projectId, agentId);
vault[key] = value;
await saveAgentVault(root, projectId, agentId, vault);
return vault;
}
/**
* Removes one vault entry; idempotent if the key doesn't exist (no write happens). Once emptied,
* the whole .vault.toml is removed. Returns the updated vault.
*/
export async function removeVaultEntry(
root: string,
projectId: string,
agentId: string,
key: string,
): Promise<Record<string, string>> {
const vault = await loadAgentVault(root, projectId, agentId);
if (!(key in vault)) return vault;
delete vault[key];
await saveAgentVault(root, projectId, agentId, vault);
return vault;
}
+48
View File
@@ -0,0 +1,48 @@
/**
* Preset content for builtin Agents; Skill documentation lives in @prismshadow/penguin-skills
* (the library files are read live when building the preset).
*
* - Every Project comes with a single builtin Agent: `default_agent` (the General Agent, the
* default conversational Agent), which has every Skill in the library installed at
* initialization. Dedicated capabilities (creating an Agent, optimizing an Agent, etc.)
* are carried by Skills rather than dedicated builtin Agents.
* - The preset carries no AGENTS.md: the default AGENTS.md is empty, with delegation and task
* conventions living in the default template's Suggested Workflows section.
* - Skill metadata is auto-injected into the system Prompt via the `{{SKILL_METADATA}}`
* placeholder; it's not registered in AGENTS.md.
*/
import { loadLibrarySkills, type LibrarySkill } from "@prismshadow/penguin-skills";
import { DEFAULT_AGENT_ID } from "./paths.js";
/** The set of Project builtin Agent ids (supplied along with the Project, cannot be deleted from Web). */
export const BUILTIN_AGENT_IDS: readonly string[] = [DEFAULT_AGENT_ID];
/** Agent initialization preset (only takes effect at initialization; ignored when loading an existing Agent). */
export interface AgentPreset {
/** Display name written to system_config.yaml. */
name?: string;
/** Description written to system_config.yaml. */
description?: string;
/** Overrides the default AGENTS.md content. */
agentsMd?: string;
/** Skills installed at initialization (installs none by default). */
skills?: LibrarySkill[];
}
/**
* The preset list for a Project's builtin Agents (each initialized in turn when the Project is
* created; an existing Agent is never overwritten). The only builtin Agent is default_agent:
* installs every Skill in the library, with no preset AGENTS.md.
*/
export function builtinProjectAgentPresets(): Array<{ agentId: string; preset: AgentPreset }> {
return [
{
agentId: DEFAULT_AGENT_ID,
preset: {
name: "General Agent",
description: "General-purpose agent that completes the user's requests with its tools.",
skills: loadLibrarySkills(),
},
},
];
}
+393
View File
@@ -0,0 +1,393 @@
/**
* Default system configuration for Agent State (written to `system_config.yaml`) and the
* default `AGENTS.md` (empty).
*
* Runtime Prompt and tool configuration should come from editable files;
* code only supplies the initial defaults. `system_config.yaml` holds the relatively stable
* system-level Prompt, built-in tools, and MCP Server configuration; `AGENTS.md` is injected
* via a system Prompt placeholder.
*
* The system Prompt is sectioned and trimmed as needed (Role/Personality/Success
* criteria/Constraints/Stop rules/File system/Suggested workflows); it does not describe
* specific tools (that comes from the tool schema). AGENTS.md, Vault/Skills, and Environment
* injection go at the end.
*
* Placeholders (`{{...}}`) appear only in the trailing injection zones (AGENTS.md / Vault /
* Skills / Environment); elsewhere the body uses angle-bracket notation such as
* \`<project_dir>\`, \`<agent_id>\`, \`<session_id>\` — these are **not substituted**; the model
* fills in the actual values from the Environment section itself.
*/
import type { MCPServerConfig, ThinkingLevelName, ToolDefinitionConfig } from "../interfaces.js";
import type { CompactionMode } from "../omnimessage/types.js";
/** Docs: /docs/configuration § "System prompt placeholders". */
export const AGENTS_MD_PLACEHOLDER = "{{AGENTS_MD}}";
export const VAULT_KEYS_PLACEHOLDER = "{{VAULT_KEYS}}";
export const SKILL_METADATA_PLACEHOLDER = "{{SKILL_METADATA}}";
export const SESSION_ID_PLACEHOLDER = "{{SESSION_ID}}";
export const CWD_PLACEHOLDER = "{{CWD}}";
export const AGENT_ID_PLACEHOLDER = "{{AGENT_ID}}";
export const PROJECT_DIR_PLACEHOLDER = "{{PROJECT_DIR}}";
export const PLATFORM_PLACEHOLDER = "{{PLATFORM}}";
export const OS_VERSION_PLACEHOLDER = "{{OS_VERSION}}";
export const DATE_PLACEHOLDER = "{{DATE}}";
/**
* Context compaction config (the `compaction` section of `system_config.yaml`).
* Docs: /docs/configuration § "Agent config".
*/
export interface CompactionConfig {
/** Context Token threshold (taken from the most recent token_usage's request.total); defaults to 128000, <=0 disables. */
max_context_length?: number;
/** Session cumulative turn threshold (counted in LLM Requests, across Tasks); defaults to -1, <=0 means no limit. */
max_session_turns?: number;
/** Compaction mode; defaults to summarize. */
mode?: CompactionMode;
/** Prompt template for summarize compaction; defaults to the built-in value (editable config, not hardcoded). */
prompt?: string;
}
/**
* System-level config for Agent State, serialized as `system_config.yaml`.
* Docs: /docs/configuration § "Agent config".
*/
export interface SystemConfig {
/** Agent display name (display name is separate from id; falls back to id when unset). */
name?: string;
/** Agent description. */
description?: string;
/** Agent State version number: a natural number, 1 on creation, incremented on successful optimization; a missing field is treated as 1. */
version?: number;
/** System-level Prompt (relatively stable; should not be modified frequently). */
system_prompt: string;
/** Max LLM turns per Task (a runtime parameter that belongs to Agent config, not specified when creating a Session). */
max_turns?: number;
model?: {
max_tokens?: number;
thinking_level?: ThinkingLevelName;
timeoutMs?: number;
};
/** Context compaction (enabled by default, max_context_length 128k, mode summarize). */
compaction?: CompactionConfig;
tools?: {
/** Built-in system tool configuration. */
builtin?: ToolDefinitionConfig[];
/** MCP Server configuration. */
mcpServers?: MCPServerConfig[];
};
}
const DEFAULT_SYSTEM_PROMPT = `# Role
You are PenguinHarness, an agent that completes the user's requests on their machine with the tools available to you.
# Personality
Communicate with the user precisely and concisely, yet with warmth. Do not repeatedly explain your tools or restate their results.
# Success criteria
- Before delivering the result, check that every problem in the request has been solved.
- Verify your work through every available means; never claim a result you did not observe.
# Constraints
- Make the smallest change that satisfies the request; do not modify unrelated files.
- Destructive operations are forbidden.
- Never kill a process you did not start yourself (e.g. to free a busy port) unless the user explicitly asks you to.
- If a tool call fails, read the error, adjust, and retry; never repeat the same failing input.
# Stop rules
- Stop and give the final answer once the success criteria are met.
- If the request is ambiguous, stop and ask the user for clarification instead of guessing their intent.
- If you hit an error you cannot resolve, stop and report the blocker to the user.
# Tool use
- Prefer solving problems with your tools: inspect the real files and environment and run real commands instead of answering from memory or guessing.
- When you need information from the internet, browse it with your shell tool — \`curl\` for pages and APIs, or Playwright (if installed) for dynamic sites.
# System markers
Some user-side messages are system-synthesized records, not user text to answer directly:
- \`<turn_aborted>\`: the previous round was interrupted. Inside are the original request, your partial thinking/text, and the tool calls already issued with their results. Continue from where it left off; do not re-run tools whose results are already included.
- \`<turn_retried>\`: the previous attempt of this round failed on a transport error (timeout or malformed response) — the user did NOT interrupt — and this request is the automatic retry. Inside are your partial thinking/text and the tool calls already executed with their results. Continue from them; do not re-run tools whose results are already included.
- \`<context_summary>\`: earlier conversation was compacted. This summary replaced the raw transcript and is its only record; treat it as established context and continue the task from it.
# File system
- Angle-bracket markers such as \`<project_dir>\`, \`<agent_id>\` and \`<session_id>\` are not literal paths — substitute the matching values from the Environment section.
- You run inside the user's working folder (\`CWD\` in Environment).
- The project directory is \`<project_dir>\`; every agent of this project lives under \`<project_dir>/agents/\`, so another agent's assets are at \`<project_dir>/agents/<its_agent_id>/agent_state/\`.
- Your own Agent State is \`<project_dir>/agents/<agent_id>/agent_state/\` — it holds your assets such as \`skills/\`, and its \`AGENTS.md\` is already included in your context. Reach these paths directly.
- For temporary and scratch files, create a subdirectory named after the current Session ID under your scratchpad: \`<project_dir>/agents/<agent_id>/scratchpad/<session_id>/\`. Build intermediates there, but always place final deliverables in the workspace (under \`CWD\`) — files left in the scratchpad are not part of your output.
- When you create or update a file in the workspace, mention its workspace-relative path in backticks (e.g. \`src/app.py\`) in your reply, so the user can open it from the message.
- Never read, copy, print or otherwise access \`<project_dir>/.project_config.toml\` or any agent's \`agent_state/.vault.toml\` — they hold the user's API keys and other secrets, which are none of your business. Configuration is CLI-only: change models or credentials with \`penguin config ...\` commands. If a task seems to require these files, say so and ask the user instead.
# Suggested workflows
These are recommendations, not requirements; adapt them as the task demands.
- For a long-horizon task, first write a plan in Markdown to \`<project_dir>/agents/<agent_id>/scratchpad/<session_id>/PLAN.md\`, containing a task overview and an itemized step-by-step plan; update it after each completed step to keep execution consistent.
- Delegate self-contained subtasks to other agents with the \`run_subagent\` tool; dispatch independent subtasks in parallel. Start every delegation prompt with your own agent id (e.g. "Caller agent: <agent_id>") and name the skill the subagent should use when the task matches one. Subagents share your Workspace — exchange data through files. If \`run_subagent\` is not in your tool list, you are the subagent: do the work yourself.
- To visit web pages, prefer Playwright when installed; otherwise \`curl\`. When building a web app or frontend, prefer React.
<developer_instructions>
Custom instructions from the developer-editable AGENTS.md.
{{AGENTS_MD}}
</developer_instructions>
# Vault
The vault holds this agent's per-agent secrets (agent_state/.vault.toml). Each entry is injected into your shell subprocesses as an environment variable — values never appear in your context. Use the variable names below in commands when a task needs them.
{{VAULT_KEYS}}
# Skills
Skills are reusable instruction packages stored under <project_dir>/agents/<agent_id>/agent_state/skills/<skill_name>/SKILL.md. There is no skill tool: when a task matches an installed skill below, or the user asks to use one (a message may start with a <use_skills> block listing skill names), first read that skill's SKILL.md in full with a shell command, then follow it. If a request only names a skill without a concrete task, ask the user what they need before starting.
{{SKILL_METADATA}}
# Environment
- Platform: {{PLATFORM}}
- OS Version: {{OS_VERSION}}
- Date: {{DATE}}
- CWD: {{CWD}}
- Agent ID: {{AGENT_ID}}
- Project Dir: {{PROJECT_DIR}}
- Session ID: {{SESSION_ID}}`;
/**
* Built-in default compaction Prompt (summarize mode): tells the model that after
* compaction the raw transcript is no longer visible and the
* summary is the only record, so it must include everything needed to continue the task,
* and no tools may be called while writing the summary.
*/
export const DEFAULT_COMPACTION_PROMPT =
"You have a partial transcript of the task above. Write a summary of it wrapped in " +
"`<summary></summary>` tags. This summary will replace the transcript: in the next " +
"context window the raw transcript above will no longer be visible and this summary " +
"will be its only record, so include everything needed to continue the task — the " +
"original request, current state, next steps, and any learnings. Do not call any " +
"tools while writing the summary; respond with text only.";
/**
* Default built-in system tools: bash execution and subagent spawning.
* Docs: /docs/tools § "Built-in tools".
*/
function defaultBuiltinTools(): ToolDefinitionConfig[] {
return [
{
name: "exec_command",
description:
"Run a shell command in the workspace to read, write, edit files and run programs. " +
"Run long-lived commands (servers, watchers, builds) in the foreground: past yield_time_ms " +
"they keep running in the background with a process_id. Do not background them with `&` — " +
"the whole process group is cleaned up when the foreground command exits.",
parameters: {
type: "object",
properties: {
cmd: {
type: "string",
description: "Shell command to execute.",
},
workdir: {
type: "string",
description:
"Working directory for the command; defaults to the cwd. Optionally a path relative to the cwd, or an absolute path.",
},
yield_time_ms: {
type: "number",
description:
"How long to wait for the command before yielding. If it is still running when this elapses, the tool returns the output so far plus a process_id, and the command keeps running in the background (drive it with input_command). Defaults to 60000; minimum 250, capped below the tool timeout.",
},
},
required: ["cmd"],
},
permission: "rw",
timeoutMs: 120000,
maxOutputLength: 16000,
},
{
name: "input_command",
description:
"Interact with a running command session started by exec_command: write to its stdin, send Ctrl-C, or poll for new output. Identify the session with its process_id.",
parameters: {
type: "object",
properties: {
process_id: {
type: "string",
description: "The process_id returned by exec_command for the running command session.",
},
chars: {
type: "string",
description:
'Characters to write to the command\'s stdin. Send "\\u0003" alone to deliver Ctrl-C (SIGINT); mixing it with other characters is an error. Empty (the default) writes nothing and only polls for new output and exit status.',
},
yield_time_ms: {
type: "number",
description:
"How long to wait for new output or exit before returning. Non-empty writes default to 250; empty polls default to 5000. Minimum 250, capped below the tool timeout.",
},
},
required: ["process_id"],
},
permission: "rw",
// An empty poll can wait out a build/test run (the yield ceiling is derived from timeoutMs, clamped inside the tool).
timeoutMs: 130000,
maxOutputLength: 16000,
},
{
name: "run_subagent",
description:
"Delegate a self-contained subtask to a subagent that runs autonomously in the same workspace and returns its final answer. Use it for focused sub-tasks you can fully specify in one prompt. Optionally choose a specific agent via `agent_id` and a model via `model_id`. " +
'Begin the prompt by identifying yourself with your own agent id (from the Environment section), e.g. "Caller agent: default_agent" — the subagent cannot otherwise tell who invoked it.',
parameters: {
type: "object",
properties: {
prompt: {
type: "string",
description:
"The complete task for the subagent: include all context it needs and the exact final output you expect back.",
},
agent_id: {
type: "string",
description:
"Which agent to run as the subagent; defaults to the current agent when omitted.",
},
model_id: {
type: "string",
description:
"Which model the subagent should use; defaults to the Project default model when omitted.",
},
yield_time_ms: {
type: "number",
description:
"How long to wait for the subagent before yielding. If it is still working when this elapses, the tool returns the output so far plus a subagent_id, and the subagent keeps running in the background (drive it with input_subagent). Defaults to 300000; minimum 250, capped below the tool timeout.",
},
},
required: ["prompt"],
},
permission: "rw",
// Subagent tasks typically run far longer than a single command, so the timeout ceiling is raised accordingly.
timeoutMs: 600000,
maxOutputLength: 16000,
},
{
name: "input_subagent",
description:
"Interact with a background subagent started by run_subagent: poll for new output, or send a follow-up prompt once it is idle to continue the same subagent session. Identify the session with its subagent_id. Pending tool approvals of the subagent are surfaced while this tool is waiting.",
parameters: {
type: "object",
properties: {
subagent_id: {
type: "string",
description: "The subagent_id returned by run_subagent for the background subagent.",
},
prompt: {
type: "string",
description:
"A follow-up task for the subagent, delivered as a new user message on the same session. Only accepted when the subagent is idle (its previous run finished). Empty (the default) sends nothing and only polls for new output and status.",
},
yield_time_ms: {
type: "number",
description:
"How long to wait for new output or completion before returning. Follow-up prompts default to 300000; empty polls default to 10000. Minimum 250, capped below the tool timeout.",
},
},
required: ["subagent_id"],
},
permission: "rw",
// Same generous timeout tier as run_subagent: an empty poll can wait a long time for the subagent to wrap up.
timeoutMs: 600000,
maxOutputLength: 16000,
},
// The image-reading tools are mutually exclusive based on the session model's type
// (marked via each entry's forModel, filtered at assembly time): read_image is designed
// for vision models (the image is fed back as image content); describe_image is designed
// for text-only models (the image plus the prompt are sent to the Project's configured
// vision model, vision_model, whose text answer becomes the tool output).
{
name: "read_image",
forModel: "vision",
description:
"Read an image and return it as image content for you to view. Accepts an http(s) URL " +
"or a local file path (relative paths resolve against the workspace). " +
"Supports png/jpeg/gif/webp up to 5MB.",
parameters: {
type: "object",
properties: {
source: {
type: "string",
description:
"Image to read: an http(s) URL, or a local file path (absolute, or relative to the workspace).",
},
},
required: ["source"],
},
permission: "r",
timeoutMs: 60000,
maxOutputLength: 16000,
},
{
name: "describe_image",
forModel: "text-only",
description:
"Describe an image and return a TEXT description of it. The current model does not accept " +
"images directly, so the image is analyzed by the project's configured vision model and " +
"you get its text answer back. Use `prompt` to ask exactly what you need to know about " +
"the image (e.g. transcribe text, describe a chart, locate a UI element). Accepts an " +
"http(s) URL or a local file path (relative paths resolve against the workspace). " +
"Supports png/jpeg/gif/webp up to 5MB.",
parameters: {
type: "object",
properties: {
source: {
type: "string",
description:
"Image to read: an http(s) URL, or a local file path (absolute, or relative to the workspace).",
},
prompt: {
type: "string",
description:
"What to ask about the image; the vision model answers this. Defaults to a detailed description.",
},
},
required: ["source"],
},
permission: "r",
// Includes one vision-model request, so the timeout is slightly wider than plain image reading.
timeoutMs: 90000,
maxOutputLength: 16000,
},
];
}
/** Agent State version number: an invalid or missing field is always treated as 1. */
export function agentStateVersion(config: Pick<SystemConfig, "version">): number {
const v = config.version;
return typeof v === "number" && Number.isInteger(v) && v >= 1 ? v : 1;
}
/** Returns the default system configuration for Agent State. */
export function defaultSystemConfig(): SystemConfig {
return {
version: 1,
system_prompt: DEFAULT_SYSTEM_PROMPT,
max_turns: 100,
model: {
max_tokens: 32000,
thinking_level: "medium",
timeoutMs: 120000,
},
compaction: {
max_context_length: 128000,
max_session_turns: -1,
mode: "summarize",
prompt: DEFAULT_COMPACTION_PROMPT,
},
tools: {
builtin: defaultBuiltinTools(),
mcpServers: [],
},
};
}
/**
* Returns the default editable `AGENTS.md` content: an empty string — no guidance is
* preprovisioned by default; Subagent delegation conventions and general task practices
* live in the default template's Suggested workflows section as a soft convention.
* Kept so initialization can still write an empty AGENTS.md file.
*/
export function defaultAgentsMd(): string {
return "";
}
@@ -0,0 +1,345 @@
/**
* Provisioning of the example Benchmark.
*
* When default_agent is initialized, it preprovisions `benchmarks/example-benchmark/`: two
* sample cases (each with statement/ and rubric/ indexed by a README.md),
* benchmark_config.toml (runs = 2), and a scoreboard.yaml with three sample evaluations —
* so the evaluation center has data out of the box. Its description states plainly that this
* is a built-in example and the whole directory can be deleted or replaced. Only
* default_agent gets this; ordinary Agents do not.
*
* Scoring numbers are self-consistent: each case's score / cost / duration_ms is the
* **average** computed from its runs array, and each evaluation's totals are the sum over
* its cases (written this way so it already satisfies the scoreboard v2 convention, and
* tests can verify it).
*/
import fs from "node:fs/promises";
import path from "node:path";
import { stringify as stringifyToml } from "smol-toml";
import { stringify as stringifyYaml } from "yaml";
import { benchmarksDir } from "./paths.js";
/** Directory name of the example Benchmark (the directory name is also its identifier). */
export const EXAMPLE_BENCHMARK_ID = "example-benchmark";
/** Contents of benchmark_config.toml (no model reference here — the model is recorded on each evaluation instead). */
const EXAMPLE_BENCHMARK_CONFIG = {
title: "Example Benchmark",
description:
"A built-in example benchmark so the evaluation charts have data out of the box. " +
"Replace it with your own.",
runs: 2,
};
/** Two sample cases: statement and scoring rubric (in English, 3-5 lines each). */
const EXAMPLE_CASES: Array<{ id: string; statement: string; rubric: string }> = [
{
id: "CASE-001-file-summary",
statement: `# Task: Summarize a project file
Read the provided \`notes.txt\` in your workspace and write \`summary.md\` containing:
1. A one-paragraph overview of at most 3 sentences.
2. A bullet list of the three most important facts.
Keep the whole summary under 150 words.
`,
rubric: `# Scoring rubric (max 5 points)
- 2 pts: \`summary.md\` exists and stays under 150 words.
- 2 pts: The three bullet facts are accurate and taken from \`notes.txt\`.
- 1 pt: The overview paragraph is coherent and at most 3 sentences.
Award partial credit per item; the case score is the sum.
`,
},
{
id: "CASE-002-data-cleanup",
statement: `# Task: Clean up a CSV dataset
The workspace contains \`users.csv\` with duplicate rows and inconsistent casing in the email column.
Produce \`users_clean.csv\` where:
1. Emails are lowercased and rows with an empty email are removed.
2. Exact duplicate rows are dropped, keeping the first occurrence.
Do not change the column order.
`,
rubric: `# Scoring rubric (max 5 points)
- 2 pts: \`users_clean.csv\` exists and keeps the original column order.
- 2 pts: Emails are lowercased, empty-email rows removed, duplicates dropped (first kept).
- 1 pt: No unrelated rows or columns were modified.
Award partial credit per item; the case score is the sum.
`,
},
];
/** Raw result of a single run (a runs element in scoreboard v2). */
interface ExampleRun {
score: number;
cost: number;
duration_ms: number;
session_id: string;
}
/**
* Raw runs for the three sample evaluations (case-level and evaluation-level metrics are
* computed from these, keeping the numbers self-consistent). Each carries the model actually
* used for that round (paired, since the evaluation center's chart splits series by model);
* the examples all use deepseek-v4-pro (a single model, single series).
*/
const EXAMPLE_EVALUATIONS: Array<{
time: string;
version: number;
provider: string;
model_id: string;
summary_title: string;
summary: string;
cases: Array<{ case: string; runs: ExampleRun[] }>;
}> = [
{
time: "2026-07-14T09:30:00Z",
version: 1,
provider: "deepseek",
model_id: "deepseek-v4-pro",
summary_title: "Baseline before any optimization",
summary:
"Example data (not a real evaluation): baseline scores of the built-in sample " +
"benchmark before any optimization. Hypothesis for the next round: the agent skips " +
"a final self-check, losing points on completeness.",
cases: [
{
case: "CASE-001-file-summary",
runs: [
{
score: 2.5,
cost: 0.012,
duration_ms: 42000,
session_id: "session-2026-07-14-09-05-11-1a2b3c01",
},
{
score: 3.5,
cost: 0.014,
duration_ms: 48000,
session_id: "session-2026-07-14-09-13-27-1a2b3c02",
},
],
},
{
case: "CASE-002-data-cleanup",
runs: [
{
score: 3.0,
cost: 0.018,
duration_ms: 66000,
session_id: "session-2026-07-14-09-21-45-1a2b3c03",
},
{
score: 3.0,
cost: 0.022,
duration_ms: 74000,
session_id: "session-2026-07-14-09-28-52-1a2b3c04",
},
],
},
],
},
{
time: "2026-07-15T09:30:00Z",
version: 2,
provider: "deepseek",
model_id: "deepseek-v4-pro",
summary_title: "Added an explicit planning step",
summary:
"Example data (not a real evaluation): after adding an explicit planning step to the " +
"system prompt (hypothesis: written plans reduce missed requirements), both cases " +
"improved. Next: tighten output formatting.",
cases: [
{
case: "CASE-001-file-summary",
runs: [
{
score: 3.5,
cost: 0.011,
duration_ms: 39000,
session_id: "session-2026-07-15-09-04-33-2b3c4d01",
},
{
score: 4.0,
cost: 0.013,
duration_ms: 45000,
session_id: "session-2026-07-15-09-12-08-2b3c4d02",
},
],
},
{
case: "CASE-002-data-cleanup",
runs: [
{
score: 3.5,
cost: 0.016,
duration_ms: 60000,
session_id: "session-2026-07-15-09-19-40-2b3c4d03",
},
{
score: 4.0,
cost: 0.02,
duration_ms: 68000,
session_id: "session-2026-07-15-09-26-59-2b3c4d04",
},
],
},
],
},
{
time: "2026-07-16T09:30:00Z",
version: 3,
provider: "deepseek",
model_id: "deepseek-v4-pro",
summary_title: "Verify deliverables before finishing",
summary:
"Example data (not a real evaluation): after instructing the agent to verify its " +
"deliverables against the statement before finishing (hypothesis: a final check " +
"catches formatting slips), scores improved again. Replace this benchmark with your " +
"own to track real progress.",
cases: [
{
case: "CASE-001-file-summary",
runs: [
{
score: 4.0,
cost: 0.01,
duration_ms: 36000,
session_id: "session-2026-07-16-09-03-21-3c4d5e01",
},
{
score: 4.5,
cost: 0.012,
duration_ms: 40000,
session_id: "session-2026-07-16-09-10-46-3c4d5e02",
},
],
},
{
case: "CASE-002-data-cleanup",
runs: [
{
score: 4.5,
cost: 0.015,
duration_ms: 55000,
session_id: "session-2026-07-16-09-18-02-3c4d5e03",
},
{
score: 4.0,
cost: 0.017,
duration_ms: 61000,
session_id: "session-2026-07-16-09-25-30-3c4d5e04",
},
],
},
],
},
];
/** Round floats to 1e-6 (so binary error from averaging/summing isn't persisted to disk). */
function round(v: number): number {
return Math.round(v * 1e6) / 1e6;
}
function average(values: number[]): number {
return round(values.reduce((a, b) => a + b, 0) / values.length);
}
function sum(values: number[]): number {
return round(values.reduce((a, b) => a + b, 0));
}
/**
* Builds the scoreboard object from raw runs data: each case's three metrics are the
* average of its runs, and each evaluation's metrics are the sum of its cases' averages
* (following the scoreboard v2 convention). Exported so tests can verify the numbers
* are self-consistent.
*/
export function buildExampleScoreboard(): {
evaluations: Array<{
time: string;
version: number;
provider: string;
model_id: string;
summary_title: string;
summary: string;
score: number;
cost: number;
duration_ms: number;
cases: Array<{
case: string;
score: number;
cost: number;
duration_ms: number;
runs: ExampleRun[];
}>;
}>;
} {
return {
evaluations: EXAMPLE_EVALUATIONS.map((e) => {
const cases = e.cases.map((c) => ({
case: c.case,
score: average(c.runs.map((r) => r.score)),
cost: average(c.runs.map((r) => r.cost)),
duration_ms: average(c.runs.map((r) => r.duration_ms)),
runs: c.runs,
}));
return {
time: e.time,
version: e.version,
provider: e.provider,
model_id: e.model_id,
summary_title: e.summary_title,
summary: e.summary,
score: sum(cases.map((c) => c.score)),
cost: sum(cases.map((c) => c.cost)),
duration_ms: sum(cases.map((c) => c.duration_ms)),
cases,
};
}),
};
}
/**
* Provisions the example Benchmark: if `benchmarks/` already exists (the user already has a
* case library), does nothing; otherwise creates `benchmarks/example-benchmark/` (config, the
* two sample cases, and the scoreboard). Callers are restricted to the default_agent
* initialization path (see agent-state.ts).
*/
export async function provisionExampleBenchmark(
root: string,
projectId: string,
agentId: string,
): Promise<void> {
const dir = benchmarksDir(root, projectId, agentId);
try {
await fs.access(dir);
return;
} catch {
// benchmarks/ does not exist: proceed with provisioning.
}
const benchDir = path.join(dir, EXAMPLE_BENCHMARK_ID);
await Promise.all(
EXAMPLE_CASES.flatMap((c) => [
fs.mkdir(path.join(benchDir, c.id, "statement"), { recursive: true }),
fs.mkdir(path.join(benchDir, c.id, "rubric"), { recursive: true }),
]),
);
await Promise.all([
fs.writeFile(
path.join(benchDir, "benchmark_config.toml"),
`${stringifyToml(EXAMPLE_BENCHMARK_CONFIG)}\n`,
"utf8",
),
fs.writeFile(
path.join(benchDir, "scoreboard.yaml"),
stringifyYaml(buildExampleScoreboard()),
"utf8",
),
...EXAMPLE_CASES.flatMap((c) => [
fs.writeFile(path.join(benchDir, c.id, "statement", "README.md"), c.statement, "utf8"),
fs.writeFile(path.join(benchDir, c.id, "rubric", "README.md"), c.rubric, "utf8"),
]),
]);
}
+16
View File
@@ -0,0 +1,16 @@
/**
* Agent State and Project config storage.
*
* Directory layout, default config, Project config read/write, Agent State load/init.
*/
export * from "./paths.js";
export * from "./default-config.js";
export * from "./builtin-agents.js";
export * from "./model-catalog.js";
export * from "./project-config.js";
export * from "./agent-state.js";
export * from "./agent-vault.js";
export * from "./example-benchmark.js";
// Skill library types and frontmatter parser (from the skills package; server reuses the same implementation via core).
export { parseSkillFrontmatter, type SkillMetadata } from "@prismshadow/penguin-skills";
+489
View File
@@ -0,0 +1,489 @@
/**
* Built-in model catalog (single source of truth): official chat models that AgentHub can
* auto-route, shared by core's default config, server's initial config, and web/cli display.
* Data verified as of 2026-07-10.
* Docs: packages/docs/content/models.{zh,en}.md (site path /docs/models) documents the
* provider groups and credential resolution described here.
*
* Three-bucket pricing convention (USD per million tokens, matching usageToTokenCounts'
* token-to-bucket mapping):
* - cache_read: the vendor's "cache hit" price;
* - cache_write: the vendor's "cache write" price (e.g. Anthropic uses 1.25 x input); vendors
* without a separate cache-write fee use the standard input price;
* - output: output price (thinking + reply).
* OpenAI charges extra for >272K input and Gemini 3.1 Pro for >200K input under official
* long-context pricing; this catalog only records the base tier (the cost center uses a
* single rate, so long-context usage will be underestimated).
*
* Scope: excludes deepseek-chat / deepseek-reasoner legacy aliases that AgentHub cannot
* auto-route (deprecated 2026-07-24), glm-5v-turbo (image input unsupported by AgentHub's GLM
* client), non-chat models (embedding / image generation / TTS), and Bedrock plus
* OpenRouter / SiliconFlow gateway mirror ids. Every model id in this catalog can be
* auto-routed by AgentHub via substring matching, so none set client_type; only custom
* OpenAI-protocol models need `client_type: "openai"`.
*
* This file imports no Node built-ins (type-only imports only), so it can be bundled directly
* for the browser.
*/
import type { ModelEntry, ModelPricing } from "./project-config.js";
/** Model provider info (used for web grouping/logo and the "API key blank falls back to env var" hint). */
export interface ModelProviderInfo {
id: string;
/** Display name (brand name, shared by Chinese and English UI). */
label: string;
/** API key env var name (AgentHub reads this automatically when credential is blank). */
envKey: string;
/** base URL env var name. */
envBaseUrlKey: string;
/** Console URL for obtaining an API key (frontend links this in the group header); none for custom. */
apiKeyUrl?: string;
/** Vendor's model list / docs page URL (frontend's "add model" dialog links this as "get model id"); none for custom. */
modelsUrl?: string;
/**
* Gateway's OpenAI-compatible endpoint (openrouter / siliconflow): used by the frontend's
* "add model" dialog to prefill base URL by group; left blank for direct vendors and custom.
*/
gatewayBaseUrl?: string;
}
/** A single built-in model's catalog entry (`modelId` is the upstream id; paired with `provider` it forms the catalog's unique key). */
export interface ModelCatalogEntry {
modelId: string;
displayName: string;
/** Provider id (one of MODEL_PROVIDERS). */
provider: string;
contextWindow?: number;
pricing?: ModelPricing;
/** Whether image input (vision modality) is supported. */
supportsVision: boolean;
/** AgentHub client protocol: required for models whose id can't be auto-routed (e.g. OpenRouter gateway models). */
clientType?: string;
/** Preset base URL (gateway models): inlined into the model entry so the user only needs to supply an API key. */
baseUrl?: string;
}
/** Each gateway's OpenAI-compatible endpoint (preset base URL for gateway models; also used as the provider's gatewayBaseUrl). */
const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
const SILICONFLOW_BASE_URL = "https://api.siliconflow.cn/v1";
/**
* Provider list (web model page groups in this order): DeepSeek first (the default model's
* provider), followed by the OpenRouter and SiliconFlow gateways, then Google Gemini before
* Anthropic; custom groups custom OpenAI-protocol models and comes last.
*/
export const MODEL_PROVIDERS: ModelProviderInfo[] = [
{
id: "deepseek",
label: "DeepSeek",
envKey: "DEEPSEEK_API_KEY",
envBaseUrlKey: "DEEPSEEK_BASE_URL",
apiKeyUrl: "https://platform.deepseek.com/api_keys",
modelsUrl: "https://api-docs.deepseek.com/quick_start/pricing",
},
// Gateways (their model ids can't be auto-routed by AgentHub, so they always use
// client_type=openai + a preset base URL): they go through AgentHub's OpenAI client, so when
// credential is blank the SDK reads **OPENAI_API_KEY / OPENAI_BASE_URL** (not the provider's
// own var names) - the env fallback hint must reflect that accurately.
{
id: "openrouter",
label: "OpenRouter",
envKey: "OPENAI_API_KEY",
envBaseUrlKey: "OPENAI_BASE_URL",
apiKeyUrl: "https://openrouter.ai/workspaces/default/keys",
modelsUrl: "https://openrouter.ai/models",
gatewayBaseUrl: OPENROUTER_BASE_URL,
},
{
id: "siliconflow",
label: "SiliconFlow",
envKey: "OPENAI_API_KEY",
envBaseUrlKey: "OPENAI_BASE_URL",
apiKeyUrl: "https://cloud.siliconflow.cn/me/account/ak",
modelsUrl: "https://cloud.siliconflow.cn/models",
gatewayBaseUrl: SILICONFLOW_BASE_URL,
},
{
id: "google",
label: "Google Gemini",
envKey: "GEMINI_API_KEY",
envBaseUrlKey: "GEMINI_BASE_URL",
apiKeyUrl: "https://aistudio.google.com/api-keys",
modelsUrl: "https://ai.google.dev/gemini-api/docs/models",
},
{
id: "anthropic",
label: "Anthropic",
envKey: "ANTHROPIC_API_KEY",
envBaseUrlKey: "ANTHROPIC_BASE_URL",
apiKeyUrl: "https://platform.claude.com/settings/keys",
modelsUrl: "https://docs.claude.com/en/docs/about-claude/models/overview",
},
{
id: "openai",
label: "OpenAI",
envKey: "OPENAI_API_KEY",
envBaseUrlKey: "OPENAI_BASE_URL",
apiKeyUrl: "https://platform.openai.com/api-keys",
modelsUrl: "https://platform.openai.com/docs/models",
},
{
id: "zhipu",
label: "Z.AI (GLM)",
envKey: "ZAI_API_KEY",
envBaseUrlKey: "ZAI_BASE_URL",
apiKeyUrl: "https://open.bigmodel.cn/apikey/platform",
modelsUrl: "https://docs.z.ai/guides/overview/pricing",
},
{
id: "moonshot",
label: "Moonshot (Kimi)",
envKey: "MOONSHOT_API_KEY",
envBaseUrlKey: "MOONSHOT_BASE_URL",
apiKeyUrl: "https://platform.kimi.com/console/api-keys",
modelsUrl: "https://platform.kimi.com/docs/pricing",
},
{ id: "custom", label: "Custom", envKey: "OPENAI_API_KEY", envBaseUrlKey: "OPENAI_BASE_URL" },
];
/** Three-bucket price literal (unit fixed to usd_per_mtok). */
/**
* Converts official CNY pricing to USD for storage (prices are always persisted in USD). The
* conversion rate matches the web display's 7:1 convention, so switching the UI to CNY shows
* exactly the vendor's official CNY price.
*/
function cny(cacheRead: number, cacheWrite: number, output: number): ModelPricing {
const r = (v: number): number => Math.round((v / 7) * 1e6) / 1e6;
return usd(r(cacheRead), r(cacheWrite), r(output));
}
function usd(cacheRead: number, cacheWrite: number, output: number): ModelPricing {
return { unit: "usd_per_mtok", cache_read: cacheRead, cache_write: cacheWrite, output };
}
/** Built-in model catalog (clustered by provider; within each provider, ordered by capability/price, highest first). */
export const MODEL_CATALOG: ModelCatalogEntry[] = [
// -- DeepSeek (official CNY pricing: cache hit / cache miss / output) --
{
modelId: "deepseek-v4-pro",
displayName: "DeepSeek V4 Pro",
provider: "deepseek",
contextWindow: 1000000,
pricing: cny(0.025, 3, 6),
supportsVision: false,
},
{
modelId: "deepseek-v4-flash",
displayName: "DeepSeek V4 Flash",
provider: "deepseek",
contextWindow: 1000000,
pricing: cny(0.02, 1, 2),
supportsVision: false,
},
// —— Anthropic ——
{
modelId: "claude-opus-4-8",
displayName: "Claude Opus 4.8",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.5, 6.25, 25),
supportsVision: true,
},
{
modelId: "claude-opus-4-7",
displayName: "Claude Opus 4.7",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.5, 6.25, 25),
supportsVision: true,
},
{
modelId: "claude-sonnet-4-6",
displayName: "Claude Sonnet 4.6",
provider: "anthropic",
contextWindow: 1000000,
pricing: usd(0.3, 3.75, 15),
supportsVision: true,
},
// —— OpenAI ——
{
modelId: "gpt-5.5",
displayName: "GPT-5.5",
provider: "openai",
contextWindow: 1050000,
pricing: usd(0.5, 5, 30),
supportsVision: true,
},
{
// No official cache discount: cache_read uses the standard input price.
modelId: "gpt-5.5-pro",
displayName: "GPT-5.5 Pro",
provider: "openai",
contextWindow: 1050000,
pricing: usd(30, 30, 180),
supportsVision: true,
},
{
modelId: "gpt-5.4",
displayName: "GPT-5.4",
provider: "openai",
contextWindow: 1050000,
pricing: usd(0.25, 2.5, 15),
supportsVision: true,
},
{
modelId: "gpt-5.4-mini",
displayName: "GPT-5.4 mini",
provider: "openai",
contextWindow: 400000,
pricing: usd(0.075, 0.75, 4.5),
supportsVision: true,
},
{
modelId: "gpt-5.4-nano",
displayName: "GPT-5.4 nano",
provider: "openai",
contextWindow: 400000,
pricing: usd(0.02, 0.2, 1.25),
supportsVision: true,
},
{
// No official cache discount: cache_read uses the standard input price.
modelId: "gpt-5.4-pro",
displayName: "GPT-5.4 Pro",
provider: "openai",
contextWindow: 1050000,
pricing: usd(30, 30, 180),
supportsVision: true,
},
// —— Google Gemini ——
{
// ≤200K input tier; >200K has official surcharge pricing (see file header comment).
modelId: "gemini-3.1-pro-preview",
displayName: "Gemini 3.1 Pro (Preview)",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.2, 2, 12),
supportsVision: true,
},
{
modelId: "gemini-3.5-flash",
displayName: "Gemini 3.5 Flash",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.15, 1.5, 9),
supportsVision: true,
},
{
modelId: "gemini-3-flash-preview",
displayName: "Gemini 3 Flash (Preview)",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.05, 0.5, 3),
supportsVision: true,
},
{
modelId: "gemini-3.1-flash-lite",
displayName: "Gemini 3.1 Flash-Lite",
provider: "google",
contextWindow: 1048576,
pricing: usd(0.025, 0.25, 1.5),
supportsVision: true,
},
// —— Z.AI (GLM) ——
{
modelId: "glm-5.2",
displayName: "GLM-5.2",
provider: "zhipu",
contextWindow: 1000000,
pricing: usd(0.26, 1.4, 4.4),
supportsVision: false,
},
{
modelId: "glm-5.1",
displayName: "GLM-5.1",
provider: "zhipu",
contextWindow: 200000,
pricing: usd(0.26, 1.4, 4.4),
supportsVision: false,
},
{
modelId: "glm-5",
displayName: "GLM-5",
provider: "zhipu",
contextWindow: 200000,
pricing: usd(0.2, 1, 3.2),
supportsVision: false,
},
// -- Moonshot (Kimi) (official CNY pricing) --
{
modelId: "kimi-k2.6",
displayName: "Kimi K2.6",
provider: "moonshot",
contextWindow: 262144,
pricing: cny(1.1, 6.5, 27),
supportsVision: true,
},
{
modelId: "kimi-k2.5",
displayName: "Kimi K2.5",
provider: "moonshot",
contextWindow: 262144,
pricing: cny(0.7, 4, 21),
supportsVision: true,
},
// -- OpenRouter (gateway: uses OpenAI-compatible protocol, preset base URL) --
{
modelId: "xiaomi/mimo-v2.5",
displayName: "MiMo-V2.5",
provider: "openrouter",
contextWindow: 1048576,
pricing: usd(0.0028, 0.14, 0.28),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
modelId: "tencent/hy3",
displayName: "Hy3",
provider: "openrouter",
contextWindow: 262144,
pricing: usd(0.035, 0.14, 0.58),
supportsVision: false,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// No official separate cache price published: cache_read uses the standard input price (no discount assumed).
modelId: "minimax/minimax-m3",
displayName: "MiniMax M3",
provider: "openrouter",
contextWindow: 1048576,
pricing: usd(0.06, 0.3, 1.2),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
{
// No official separate cache price published: cache_read uses the standard input price.
modelId: "stepfun/step-3.7-flash",
displayName: "Step 3.7 Flash",
provider: "openrouter",
contextWindow: 256000,
pricing: usd(0.04, 0.2, 1.15),
supportsVision: true,
clientType: "openai",
baseUrl: OPENROUTER_BASE_URL,
},
// -- SiliconFlow (gateway, official CNY pricing: cache hit / input / output) --
{
modelId: "zai-org/GLM-5.2",
displayName: "GLM-5.2",
provider: "siliconflow",
contextWindow: 1000000,
pricing: cny(2, 8, 28),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "deepseek-ai/DeepSeek-V4-Pro",
displayName: "DeepSeek V4 Pro",
provider: "siliconflow",
contextWindow: 1000000,
pricing: cny(0.1, 12, 24),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
{
modelId: "meituan-longcat/LongCat-2.0",
displayName: "LongCat 2.0",
provider: "siliconflow",
contextWindow: 1000000,
pricing: cny(0.1, 5, 20),
supportsVision: false,
clientType: "openai",
baseUrl: SILICONFLOW_BASE_URL,
},
];
/** Looks up a catalog entry by (provider, upstream id) pair (**the sole catalog-matching entry point**); returns undefined if not in the catalog. */
export function catalogEntryFor(
provider: string,
upstreamId: string,
): ModelCatalogEntry | undefined {
return MODEL_CATALOG.find((m) => m.provider === provider && m.modelId === upstreamId);
}
/**
* Infers the provider for an upstream id from the built-in catalog (used to default
* `provider` on `model add`): if it matches a catalog entry, use that entry's provider
* (upstream ids are globally unique within the catalog); otherwise custom.
*/
export function inferProviderForUpstream(upstreamId: string): string {
return MODEL_CATALOG.find((m) => m.modelId === upstreamId)?.provider ?? "custom";
}
/** Looks up provider info by provider id; returns undefined for an unknown id. */
export function providerInfo(providerId: string): ModelProviderInfo | undefined {
return MODEL_PROVIDERS.find((p) => p.id === providerId);
}
/** Env var fallback for a single model (the var names AgentHub's client actually reads when api_key / base_url is blank). */
export interface ModelEnvInfo {
envKey: string;
envBaseUrlKey: string;
}
/**
* Resolves the env var fallback for a model: mirrors AgentHub's
* AutoLLMClient routing rules (verified against agenthub v0.3.3 autoClient.ts) - an explicit
* client_type takes priority, otherwise routes to a client by lowercase substring match on
* model_id, returning the var pair that client reads; branch order matches AutoLLMClient.
* Returns undefined on no match (AgentHub will reject that id: it needs an explicit
* client_type, or should be added under custom / a self-built group via the OpenAI protocol).
*/
export function resolveModelEnv(modelId: string, clientType?: string): ModelEnvInfo | undefined {
const t = (clientType || modelId).toLowerCase();
const env = (prefix: string): ModelEnvInfo => ({
envKey: `${prefix}_API_KEY`,
envBaseUrlKey: `${prefix}_BASE_URL`,
});
if (t.includes("gemini-3") || t.includes("gemini-embedding")) return env("GEMINI");
if (
t.includes("claude") &&
(t.includes("4-7") || t.includes("4-8") || t.includes("-5") || t.includes("4-6"))
) {
return env("ANTHROPIC");
}
if (t.includes("gpt-5.4") || t.includes("gpt-5.5")) return env("OPENAI");
if (t.includes("glm-5")) return env("ZAI");
if (t.includes("kimi-k2.5") || t.includes("kimi-k2.6")) return env("MOONSHOT");
if (t.includes("deepseek-v4")) return env("DEEPSEEK");
if (t.includes("openai")) return env("OPENAI");
return undefined;
}
/**
* Catalog -> preset ModelEntry list (shared by defaultProjectConfig and the server's initial
* config, avoiding duplicate hand-written copies). `provider` and `model_id` are persisted as
* separate fields (`model_id` is the plain upstream id); models whose upstream id can be
* auto-routed by AgentHub leave client_type unset; gateway models (OpenRouter / SiliconFlow)
* explicitly set client_type=openai and inline a preset base_url (no secrets included, so the
* user only needs to supply an API key).
*/
export function presetModelEntries(): ModelEntry[] {
return MODEL_CATALOG.map((m) => ({
provider: m.provider,
model_id: m.modelId,
...(m.contextWindow !== undefined ? { context_window: m.contextWindow } : {}),
...(m.clientType !== undefined ? { client_type: m.clientType } : {}),
...(m.pricing ? { pricing: { ...m.pricing } } : {}),
// ModelEntry.vision defaults to supported: only models that don't support images
// explicitly persist false (drives the read_image / describe_image choice and input
// image hand-off, see project-config.ts).
...(m.supportsVision ? {} : { vision: false }),
...(m.baseUrl !== undefined ? { base_url: m.baseUrl } : {}),
}));
}
+114
View File
@@ -0,0 +1,114 @@
/**
* Local directory layout for Agent State and Project config.
*
* Strictly follows the `~/.penguin/data/<project>/agents/<agent>/...` structure.
* This module only provides constants and pure path functions; it never creates directories or reads/writes files.
* Docs: /docs/sessions-and-traces § "Data layout".
*/
import os from "node:os";
import path from "node:path";
/** Default Project id used when none is specified. */
export const DEFAULT_PROJECT_ID = "default_project";
/** Default Agent id used when none is specified. */
export const DEFAULT_AGENT_ID = "default_agent";
/**
* Resolves the local data root directory.
* Prefers the `PENGUIN_HOME` environment variable, otherwise falls back to `~/.penguin/data`
* (under the hidden `~/.penguin` home so it never collides with unrelated folders, and in a
* `data/` subdir kept separate from the installer's binaries under `~/.penguin`).
*/
export function resolveRoot(): string {
return process.env.PENGUIN_HOME ?? path.join(os.homedir(), ".penguin", "data");
}
/** `<root>/<projectId>`. */
export function projectDir(root: string, projectId: string): string {
return path.join(root, projectId);
}
/** `<projectDir>/agents`, the container directory holding every Agent in the Project. */
export function agentsDir(root: string, projectId: string): string {
return path.join(projectDir(root, projectId), "agents");
}
/** `<projectDir>/agents/<agentId>`. */
export function agentDir(root: string, projectId: string, agentId: string): string {
return path.join(agentsDir(root, projectId), agentId);
}
/** `<agentDir>/agent_state`. */
export function agentStateDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "agent_state");
}
/** `<agentDir>/traces`. */
export function tracesDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "traces");
}
/** `<agentDir>/scratchpad`, the Agent's temporary/draft file directory (the model creates a subdirectory per Session id). */
export function scratchpadDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "scratchpad");
}
/** `<agentDir>/workspaces`. */
export function workspacesDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "workspaces");
}
/**
* `<projectDir>/.project_config.toml`, the Project's single config file (a hidden file, not
* shown by default `ls`, written with mode 0600; model entries are inlined with their credential,
* see state/project-config.ts).
*/
export function projectConfigPath(root: string, projectId: string): string {
return path.join(projectDir(root, projectId), ".project_config.toml");
}
/** `<agentStateDir>/system_config.yaml`. */
export function systemConfigPath(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "system_config.yaml");
}
/** `<agentStateDir>/AGENTS.md`. */
export function agentsMdPath(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "AGENTS.md");
}
/** `<agentStateDir>/.vault.toml`, the Agent-level environment-variable vault (see state/agent-vault.ts). */
export function agentVaultPath(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), ".vault.toml");
}
/** `<agentStateDir>/tools`, reserved for user-defined Tool config. */
export function toolsDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "tools");
}
/** `<agentStateDir>/memory`. */
export function memoryDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "memory");
}
/** `<agentStateDir>/skills`. */
export function skillsDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "skills");
}
/** `<agentStateDir>/schedule`, the scheduled-task directory (doesn't exist when unconfigured). */
export function scheduleDir(root: string, projectId: string, agentId: string): string {
return path.join(agentStateDir(root, projectId, agentId), "schedule");
}
/** `<agentDir>/benchmarks`, the capability-evaluation question bank and scores (doesn't exist when unconfigured). */
export function benchmarksDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "benchmarks");
}
/** `<agentDir>/snapshots`, Agent State version snapshots (doesn't exist when unconfigured). */
export function snapshotsDir(root: string, projectId: string, agentId: string): string {
return path.join(agentDir(root, projectId, agentId), "snapshots");
}
+432
View File
@@ -0,0 +1,432 @@
/**
* Project config storage (`<project>/.project_config.toml`).
*
* Records the available Models, the default Model, and each Model's credential (Model
* is decoupled from Agent — the Model selection isn't stored in Agent State, but maintained by
* the Project). Config is persisted as TOML.
*
* `.project_config.toml` is the Project's **single config file**: a hidden file (not shown by
* default `ls`), written to disk with mode 0600; credentials (api_key / base_url) are **inlined
* on the model entry** rather than split into a supplementary area and a separate secrets file.
* It can only be read/written via the system interfaces (CLI / Web) — never hand-edited by the
* model or the user; the system Prompt is forbidden from reading this file, `loadProjectConfig`
* returns plaintext, and masking is applied at the interface layer (when shown by server / cli).
*
* Model references are **fully split into separate fields**: an entry stores
* `provider` and `model_id` as two independent fields, with the `(provider, model_id)` pair as
* the unique key — string concatenation like `<provider>/<id>` is forbidden anywhere in the
* pipeline. `model_id` is the upstream request id, sent to AgentHub unchanged; `default_model` /
* `vision_model` are paired `{ provider, model_id }` references (a TOML inline table).
*/
import fs from "node:fs/promises";
import path from "node:path";
import { parse as parseToml, stringify as stringifyToml } from "smol-toml";
import { inferProviderForUpstream, presetModelEntries } from "./model-catalog.js";
import { projectConfigPath } from "./paths.js";
/** Model reference: a `(provider, model_id)` pair (never string-concatenated anywhere). */
export interface ModelRef {
provider: string;
/** Upstream model id (the request id sent to AgentHub unchanged). */
model_id: string;
}
/**
* Display form of a paired reference (shared by error messages and CLI output):
* `(provider=..., model_id=...)`. For display only — it isn't any storage or addressing format.
*/
export function formatModelRef(ref: ModelRef): string {
return `(provider=${ref.provider}, model_id=${ref.model_id})`;
}
/**
* Pricing for a single Model: three price buckets, in USD per million tokens.
* Docs: /docs/configuration § "Project config".
*/
export interface ModelPricing {
/** Pricing unit tag; currently only `usd_per_mtok` (USD per million tokens). */
unit: "usd_per_mtok";
cache_read: number;
cache_write: number;
output: number;
}
/**
* A single available Model entry (credential inlined, single config file).
* Docs: /docs/models § "The per-Project model table".
*/
export interface ModelEntry {
/** provider group (stored separately from `model_id`; the pair is the entry's unique key). */
provider: string;
/** Upstream model id: the actual request id sent to AgentHub, used paired with provider for display, pricing, and stats. */
model_id: string;
context_window?: number;
/**
* AgentHub client protocol (`openai` / `claude-4-8` / `deepseek-v4` / …); defaults to being
* inferred by AgentHub from the request id (`model_id`). A third-party model speaking the
* OpenAI protocol should set this to `openai`.
*/
client_type?: string;
/**
* Display name (the model page card title): only persisted when it differs from the builtin
* catalog (the user renamed it / a custom model); when not persisted, it's inferred from the
* builtin catalog by `(provider, model_id)`, falling back to displaying model_id if it can't be
* inferred.
*/
display_name?: string;
/**
* Whether image input is supported (vision/multimodal); defaults to supported. For a model
* tagged `false` (e.g. DeepSeek): images from conversation input are saved to the session
* scratchpad and handed over as a file path spliced into the text, and the image-reading tool
* switches to describe_image (a vision model reads on its behalf) — the image never directly
* enters that session's history.
*/
vision?: boolean;
/** Pricing info; absent means this Model's cost isn't counted. */
pricing?: ModelPricing;
/** API key (inlined credential); left empty falls back to the vendor's environment variable. */
api_key?: string;
/** Custom base URL (inlined credential); preset for gateway models. */
base_url?: string;
/** api_key's write timestamp (ISO 8601; a display field maintained by the interface layer). */
created_at?: string;
}
/**
* Project-level config.
* Docs: /docs/configuration § "Project config".
*/
export interface ProjectConfig {
/** Project display name (the display name is separate from the id, shown as the id when unset). */
name?: string;
/** Paired reference to the default Model; must point to an entry in `models`. */
default_model?: ModelRef;
/**
* The vision model used by read_image to read on behalf of a session model (when a session
* model with `vision=false` reads an image, it's handed to this model to describe and the tool
* returns text); must point to an entry in `models` (a paired reference). Unconfigured by
* default — models that don't support images won't be able to read images.
*/
vision_model?: ModelRef;
models: ModelEntry[];
}
/**
* Returns the Project's default config: every entry from the preset builtin model catalog
* (including context_window / pricing / vision tags and the preset base_url for gateway models,
* with no keys included) — the user only needs to fill in an API key as needed (left empty falls
* back to the vendor's environment variable).
*/
export function defaultProjectConfig(): ProjectConfig {
return {
default_model: { provider: "deepseek", model_id: "deepseek-v4-pro" },
models: presetModelEntries(),
};
}
/** The old format (concatenated storage id / string reference) is never migrated: reading it reports a clear error immediately (the product hasn't shipped yet). */
const OLD_FORMAT_HINT =
"产品未发布不做迁移:请删除该配置文件后用 `penguin config model add/default` 重建。";
/** Validates the default_model / vision_model fields: must be a { provider, model_id } paired reference. */
function parseRefField(file: string, name: string, value: unknown): ModelRef | undefined {
if (value === undefined) return undefined;
const ref = value as { provider?: unknown; model_id?: unknown };
if (
typeof value !== "object" ||
value === null ||
typeof ref.provider !== "string" ||
typeof ref.model_id !== "string"
) {
throw new Error(
`.project_config.toml 的 ${name} 是旧版本/非法格式(须为 { provider = "...", model_id = "..." } 成对引用):${file}。${OLD_FORMAT_HINT}`,
);
}
return { provider: ref.provider, model_id: ref.model_id };
}
/** Validates a model entry: both provider and model_id must be strings (an old-format entry is missing provider). */
function assertModelEntry(file: string, entry: unknown): ModelEntry {
const m = entry as { provider?: unknown; model_id?: unknown };
if (
typeof entry !== "object" ||
entry === null ||
typeof m.provider !== "string" ||
typeof m.model_id !== "string"
) {
throw new Error(
`.project_config.toml 的 models 条目是旧版本/非法格式(provider 与 model_id 须为两个独立字段):${file}。${OLD_FORMAT_HINT}`,
);
}
return entry as ModelEntry;
}
/**
* Loads the Project config; returns the default config (without writing to disk) if
* `.project_config.toml` doesn't exist. Returns plaintext (masking is applied at the interface
* layer); reports a clear error when the old format (a string reference / an entry missing
* provider) is read.
*/
export async function loadProjectConfig(root: string, projectId: string): Promise<ProjectConfig> {
const file = projectConfigPath(root, projectId);
let raw: string;
try {
raw = await fs.readFile(file, "utf8");
} catch (err) {
if ((err as NodeJS.ErrnoException).code === "ENOENT") return defaultProjectConfig();
throw err;
}
// Defensive: parseToml may return null/undefined for an empty file, and destructuring it would throw a TypeError.
const parsed = (parseToml(raw) ?? {}) as Record<string, unknown>;
const defaultModel = parseRefField(file, "default_model", parsed.default_model);
const visionModel = parseRefField(file, "vision_model", parsed.vision_model);
return {
...(parsed.name !== undefined ? { name: parsed.name as string } : {}),
...(defaultModel !== undefined ? { default_model: defaultModel } : {}),
...(visionModel !== undefined ? { vision_model: visionModel } : {}),
models: ((parsed.models as unknown[] | undefined) ?? []).map((m) => assertModelEntry(file, m)),
};
}
/** A TOML inline table for a paired reference (reuses smol-toml's string serialization, guaranteeing correct escaping). */
function tomlInlineRef(ref: ModelRef): string {
const kv = (obj: Record<string, string>): string => stringifyToml(obj).trim();
return `{ ${kv({ provider: ref.provider })}, ${kv({ model_id: ref.model_id })} }`;
}
/** Whether a value has the paired-reference shape ({ provider, model_id }, two string fields). */
function isModelRefShape(v: unknown): v is ModelRef {
if (v === null || typeof v !== "object" || Array.isArray(v)) return false;
const o = v as Record<string, unknown>;
return typeof o.provider === "string" && typeof o.model_id === "string";
}
/**
* Renders the full text of `.project_config.toml` — the **single source of the write format
* site-wide** (shared by core's saveProjectConfig and the interface layer's full-table write, to
* avoid the same file ending up in two different formats).
*
* Paired references (default_model / vision_model) are rendered as a TOML inline table
* `{ provider = "...", model_id = "..." }`; `models` is always
* placed last, since any table header after `[[models]]` would be read as its sub-table. Unknown
* extension fields are kept as-is.
*/
export function renderProjectConfigToml(data: Record<string, unknown>): string {
const head: string[] = [];
for (const [key, value] of Object.entries(data)) {
if (value === undefined || key === "models") continue;
head.push(
isModelRefShape(value)
? `${key} = ${tomlInlineRef(value)}`
: stringifyToml({ [key]: value }).trim(),
);
}
const models = Array.isArray(data.models) ? data.models : [];
return [...head, stringifyToml({ models })].join("\n");
}
/**
* Saves the Project config: writes the full table to the single config file
* `.project_config.toml`. The file contains secrets like api_key, so it's written to disk with
* mode 0600 (a hidden file blocks `ls`, not reads; mode only takes effect on creation, so chmod
* converges an existing file too).
*/
export async function saveProjectConfig(
root: string,
projectId: string,
cfg: ProjectConfig,
): Promise<void> {
const file = projectConfigPath(root, projectId);
await fs.mkdir(path.dirname(file), { recursive: true });
await fs.writeFile(file, renderProjectConfigToml({ ...cfg }), {
encoding: "utf8",
mode: 0o600,
});
await fs.chmod(file, 0o600);
}
/**
* Adds or updates a Model:
* - Upserts into `models`, deduplicated by the `(provider, model_id)` pair (provider may be
* omitted — the builtin catalog is used to infer the upstream id's group, falling back to
* custom if it can't be inferred);
* - If `api_key`/`base_url` are provided, they're written inline into the entry;
* - Set as the default Model (a paired reference) when `opts.setDefault` is true.
* Reads the existing config (or the default), saves after the change, and returns the updated
* config.
*/
export async function addModel(
root: string,
projectId: string,
entry: {
/** provider group; inferred from the builtin catalog when omitted (`inferProviderForUpstream`, falling back to custom if it can't be inferred). */
provider?: string;
/** Upstream model id (sent to AgentHub unchanged). */
model_id: string;
context_window?: number;
client_type?: string;
/** Whether image input is supported (vision/multimodal); keeps the existing value by default (treated as supported if never set). */
vision?: boolean;
/** Price input may cover only some buckets; merged and written as a complete `ModelPricing`. */
pricing?: Partial<ModelPricing>;
api_key?: string;
base_url?: string;
},
opts?: { setDefault?: boolean },
): Promise<ProjectConfig> {
const cfg = await loadProjectConfig(root, projectId);
const provider = entry.provider ?? inferProviderForUpstream(entry.model_id);
// upsert: layers new fields on top of the existing entry; fields not explicitly provided
// (e.g. context_window) keep their existing value, so a call like "just add an api_key"
// doesn't wipe out the prior config.
const idx = cfg.models.findIndex((m) => m.provider === provider && m.model_id === entry.model_id);
const existing = idx >= 0 ? cfg.models[idx] : undefined;
const modelEntry: ModelEntry = {
provider,
model_id: entry.model_id,
};
const contextWindow = entry.context_window ?? existing?.context_window;
if (contextWindow !== undefined) {
modelEntry.context_window = contextWindow;
}
const clientType = entry.client_type ?? existing?.client_type;
if (clientType !== undefined) {
modelEntry.client_type = clientType;
}
// The display name and api_key write timestamp are not set by this function; kept as-is on upsert.
if (existing?.display_name !== undefined) {
modelEntry.display_name = existing.display_name;
}
const vision = entry.vision ?? existing?.vision;
if (vision !== undefined) {
modelEntry.vision = vision;
}
// The three price buckets are merged field by field: an unspecified bucket keeps its existing
// value (the same policy as context_window/credential); the unit is fixed to usd_per_mtok, and
// the complete pricing is written as long as any bucket is present.
const mergedPricing: Partial<ModelPricing> = {
...existing?.pricing,
...entry.pricing,
};
if (
mergedPricing.cache_read !== undefined ||
mergedPricing.cache_write !== undefined ||
mergedPricing.output !== undefined
) {
modelEntry.pricing = {
unit: "usd_per_mtok",
cache_read: mergedPricing.cache_read ?? 0,
cache_write: mergedPricing.cache_write ?? 0,
output: mergedPricing.output ?? 0,
};
}
// Inline credential entry: fields not provided keep their existing value.
const apiKey = entry.api_key ?? existing?.api_key;
if (apiKey !== undefined) {
modelEntry.api_key = apiKey;
}
const baseUrl = entry.base_url ?? existing?.base_url;
if (baseUrl !== undefined) {
modelEntry.base_url = baseUrl;
}
if (existing?.created_at !== undefined) {
modelEntry.created_at = existing.created_at;
}
if (idx >= 0) {
cfg.models[idx] = modelEntry;
} else {
cfg.models.push(modelEntry);
}
if (opts?.setDefault) {
cfg.default_model = { provider, model_id: entry.model_id };
}
await saveProjectConfig(root, projectId, cfg);
return cfg;
}
/**
* Sets the default Model and saves. The target reference must exist in `models` (a reference
* pointing outside the config would make createSession error immediately); throws otherwise.
*/
export async function setDefaultModel(
root: string,
projectId: string,
ref: ModelRef,
): Promise<ProjectConfig> {
const cfg = await loadProjectConfig(root, projectId);
if (!getModel(cfg, ref)) {
throw new Error(
`default_model 必须指向已配置的模型:${formatModelRef(ref)} 不在 models 中。用 \`penguin config model list\` 查看已配置的模型。`,
);
}
cfg.default_model = { provider: ref.provider, model_id: ref.model_id };
await saveProjectConfig(root, projectId, cfg);
return cfg;
}
/**
* Sets the vision model used to read images on behalf of read_image, and saves. The target
* reference must exist in `models` and not be tagged `vision=false` (a model that doesn't support
* images can't read on someone's behalf); throws otherwise.
*/
export async function setVisionModel(
root: string,
projectId: string,
ref: ModelRef,
): Promise<ProjectConfig> {
const cfg = await loadProjectConfig(root, projectId);
const entry = getModel(cfg, ref);
if (!entry) {
throw new Error(
`vision_model 必须指向已配置的模型:${formatModelRef(ref)} 不在 models 中。用 \`penguin config model list\` 查看已配置的模型。`,
);
}
if (entry.vision === false) {
throw new Error(`vision_model 不能指向标注为不支持图片的模型:${formatModelRef(ref)}。`);
}
cfg.vision_model = { provider: ref.provider, model_id: ref.model_id };
await saveProjectConfig(root, projectId, cfg);
return cfg;
}
/** Looks up a Model entry exactly by its `(provider, model_id)` paired reference; returns `undefined` if it doesn't exist. */
export function getModel(cfg: ProjectConfig, ref: ModelRef): ModelEntry | undefined {
return cfg.models.find((m) => m.provider === ref.provider && m.model_id === ref.model_id);
}
/**
* Resolves a model reference (the **single entry point for "provider omitted"**, shared by core
* and CLI/server — never set up a second one):
* - `provider` given: validated for existence by exact paired reference;
* - `provider` omitted: an exact-match lookup on `model_id` (no fuzzy matching of any kind) —
* resolvable only when **exactly one** entry matches; 0 or multiple matches always report a
* clear error (an ambiguity error lists the candidate paired references).
*/
export function resolveModelRef(cfg: ProjectConfig, modelId: string, provider?: string): ModelRef {
if (provider !== undefined) {
const ref: ModelRef = { provider, model_id: modelId };
if (!getModel(cfg, ref)) {
throw new Error(
`Model 不在 Project 配置中:${formatModelRef(ref)}。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`,
);
}
return ref;
}
const candidates = cfg.models.filter((m) => m.model_id === modelId);
if (candidates.length === 1) {
return { provider: candidates[0]!.provider, model_id: modelId };
}
if (candidates.length === 0) {
throw new Error(
`Model 不在 Project 配置中:没有 model_id 为 ${modelId} 的条目。请用 \`penguin config model list\` 查看已配置的模型,或用 \`penguin config model add\` 添加。`,
);
}
throw new Error(
`模型引用有歧义:model_id ${modelId} 命中多个条目——${candidates
.map((m) => formatModelRef({ provider: m.provider, model_id: m.model_id }))
.join("、")}。请补充 provider 以给出成对引用。`,
);
}
+10
View File
@@ -0,0 +1,10 @@
export { Writer, readTrace } from "./writer.js";
export type { WriterOptions } from "./writer.js";
export {
findLatestTraceFile,
latestSessionId,
parseTraceLines,
readTraceTolerant,
resumeTrace,
} from "./resume.js";
export type { LocatedTraceFile, ResumeResult } from "./resume.js";
+449
View File
@@ -0,0 +1,449 @@
/**
* Trace replay — the core of Session resume.
*
* Replay produces two results: the **history** injected via setHistory (committed turns only),
* and the **carry-over** input resent with the first `run` after resume. Resume is **best-effort**:
* Trace only records real messages, so synthesized carry-over (`<turn_aborted>` flattening, pairing
* placeholders) is never written to Trace — replay reconstructs from the original messages
* (unanswered input is resent as-is, pairing placeholders are resynthesized as needed). History is
* guaranteed to be **structurally valid** (turns complete, tool_call pairs matched), not a
* byte-for-byte match of what AgentHub actually received; incomplete model output (thinking/text)
* is allowed to be lost.
*
* Messages are attributed to a Request by **position**, not by content inspection:
* - Input = user-side messages accumulated after the previous `request` `stop` (the first
* Request is `session_meta`) and before this `start` (messages are written to Trace before
* being sent with the request); user-side messages that land between `start` and `stop`
* (output from parallel tools completing during the request) count toward the **next** turn's
* input.
* - Output = assistant messages between this `start` and `stop`.
*
* Determination order: first check file-level compaction closure; then evaluate turn by turn
* (completed turns go to history, others are dropped wholesale while keeping outputs paired with
* already-committed tool_calls); finally, the remaining input is the carry-over, with pairing
* backfill applied.
* Docs: /docs/sessions-and-traces § "Session recovery".
*/
import { readdir, readFile } from "node:fs/promises";
import { join } from "node:path";
import {
emptyTokenCounts,
isCompleteModelMessage,
isEventMessage,
isSessionMeta,
toolCallOutput,
userText,
} from "../omnimessage/index.js";
import type {
CompactionEndPayload,
CompleteModelMessage,
OmniMessage,
RequestBeginPayload,
RequestEndPayload,
SessionMetaMessage,
TokenCounts,
TokenUsagePayload,
ToolCallPayload,
} from "../omnimessage/index.js";
import { extractSummary } from "../engine/context-engine.js";
/** Replay result: all the state needed to resume a Session. */
export interface ResumeResult {
/** Committed history (complete model_msg, in order), injected in one shot via setHistory; empty on compaction closure. */
history: CompleteModelMessage[];
/**
* Pending input (carry-over): resent alongside new input with the first `run` after resume.
* Already includes pairing-backfill placeholders — placeholders exist only in memory
* (synthesized carry-over is never written to Trace) and are resynthesized on each resume.
*/
carryOver: OmniMessage[];
/** Compaction closure (file-level): this file's context is fully closed; resume starts a new, empty context. */
contextClosed: boolean;
/** Compaction closure in summarize mode: the reconstructed `<context_summary>` summary, prepended to the next run's input. */
pendingSummary?: OmniMessage;
/** Session-level cumulative Token carry-over (the session value from the last token_usage). */
sessionTokens: TokenCounts;
/** The request.total from the last token_usage (context usage figure). */
lastRequestTotal: number;
/** Session cumulative turn count carry-over (count of completed requests). */
sessionTurns: number;
/** Rendering view: this context's complete model_msg plus key event_msg entries (including interrupted turns and their markers); empty on compaction closure. */
renderMessages: OmniMessage[];
/** The file's first session_meta; null if missing (unresumable — the caller reports the error). */
meta: SessionMetaMessage | null;
}
/** Content of the pairing-backfill placeholder output (the tool hadn't finished and no output was persisted before the process exited). */
const PROCESS_EXIT_PLACEHOLDER = "[interrupted: process exited before the tool finished]";
/**
* Parse Trace JSONL content. Tolerates a **truncated last line** left behind by an abnormal
* process exit (that line is ignored); corruption in the middle is outside the crash window
* (append-only, single writer), so it throws loudly.
*/
export function parseTraceLines(content: string): OmniMessage[] {
const lines = content.split("\n");
const out: OmniMessage[] = [];
for (let i = 0; i < lines.length; i++) {
const line = lines[i]!.trim();
if (!line) continue;
try {
out.push(JSON.parse(line) as OmniMessage);
} catch (err) {
const isLastNonEmpty = lines.slice(i + 1).every((l) => l.trim().length === 0);
if (isLastNonEmpty) break;
throw err;
}
}
return out;
}
/** Read and parse a Trace file (tolerates a truncated last line). */
export async function readTraceTolerant(path: string): Promise<OmniMessage[]> {
return parseTraceLines(await readFile(path, "utf8"));
}
/** A located Trace file: its path, containing date-directory name, and index. */
export interface LocatedTraceFile {
path: string;
dateDir: string;
index: number;
}
const TRACE_FILE_RE = /^(.+)_(\d{3})\.jsonl$/;
/**
* Locate the **highest-index** Trace file for a Session (one Trace file corresponds to one
* complete model context). Scans `<tracesDir>/<yyyy-mm-dd>/<sessionId>_<index3>.jsonl`; returns
* null if no match is found.
*/
export async function findLatestTraceFile(
tracesDir: string,
sessionId: string,
): Promise<LocatedTraceFile | null> {
let best: LocatedTraceFile | null = null;
for (const dateDir of await listDirs(tracesDir)) {
for (const file of await listFiles(join(tracesDir, dateDir))) {
const match = TRACE_FILE_RE.exec(file);
if (!match || match[1] !== sessionId) continue;
const index = Number(match[2]);
if (!best || index > best.index) {
best = { path: join(tracesDir, dateDir, file), dateDir, index };
}
}
}
return best;
}
/**
* The id of the most recent Session under this Agent, determined by the timestamp embedded in
* session_id (ids are zero-padded, so lexical order equals chronological order). Returns null if
* there are no Sessions.
*/
export async function latestSessionId(tracesDir: string): Promise<string | null> {
const dateDirs = (await listDirs(tracesDir)).sort((a, b) => b.localeCompare(a));
for (const dateDir of dateDirs) {
const files = (await listFiles(join(tracesDir, dateDir))).sort((a, b) => b.localeCompare(a));
for (const file of files) {
const match = TRACE_FILE_RE.exec(file);
if (!match) continue;
if (await hasResumableTraceContent(join(tracesDir, dateDir, file))) {
return match[1]!;
}
}
}
return null;
}
async function hasResumableTraceContent(file: string): Promise<boolean> {
try {
const messages = await readTraceTolerant(file);
return messages.some((msg) => !isSessionMeta(msg));
} catch {
return false;
}
}
async function listDirs(dir: string): Promise<string[]> {
try {
const entries = await readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isDirectory()).map((e) => e.name);
} catch {
return []; // traces directory doesn't exist yet: no Sessions
}
}
async function listFiles(dir: string): Promise<string[]> {
try {
const entries = await readdir(dir, { withFileTypes: true });
return entries.filter((e) => e.isFile()).map((e) => e.name);
} catch {
return [];
}
}
function isRequestBegin(msg: OmniMessage): msg is OmniMessage<RequestBeginPayload> {
return isEventMessage(msg) && (msg.payload as { type?: string }).type === "request_begin";
}
function isRequestEnd(msg: OmniMessage): msg is OmniMessage<RequestEndPayload> {
return isEventMessage(msg) && (msg.payload as { type?: string }).type === "request_end";
}
function isCompactionEnd(msg: OmniMessage): msg is OmniMessage<CompactionEndPayload> {
return isEventMessage(msg) && (msg.payload as { type?: string }).type === "compaction_end";
}
function toolCallOutputId(msg: OmniMessage): string | null {
const p = msg.payload as { type?: string; tool_call_id?: string };
return p.type === "tool_call_output" ? (p.tool_call_id ?? null) : null;
}
/**
* Replay a Trace file (the current context), reconstructing history and carry-over input.
* Input is the message sequence parsed by `readTraceTolerant`.
*/
export function resumeTrace(messages: OmniMessage[]): ResumeResult {
const meta = (messages.find(isSessionMeta) as SessionMetaMessage | undefined) ?? null;
const sessionTokens = lastSessionTokens(messages);
const lastRequestTotal = lastRequestTotalOf(messages);
// —— First check the file-level case: compaction closure (the file ends with a completed
// compaction stop and no new file was opened, i.e. it's still the latest index at resume time)
// — this file's context is fully closed, so the whole file is not replayed.
const last = messages[messages.length - 1];
if (last && isCompactionEnd(last)) {
const p = last.payload;
if (p.status === "completed") {
const result: ResumeResult = {
history: [],
carryOver: [],
contextClosed: true,
sessionTokens,
lastRequestTotal: 0, // new context has no usage yet
sessionTurns: 0, // turn count resets after compaction completes
renderMessages: [],
meta,
};
if (p.mode === "summarize") {
// Reconstruct the summary from the compaction request's output (the assistant text of
// the last completed Request).
const summaryText = lastCompletedRequestText(messages);
result.pendingSummary = userText(
`<context_summary>\n${extractSummary(summaryText)}\n</context_summary>`,
);
}
return result;
}
}
// —— Turn-by-turn determination + pending-input convergence.
const history: CompleteModelMessage[] = [];
/** Pending-input buffer: user-side messages not yet sent with any committed Request. */
let pending: OmniMessage[] = [];
/** The current Request's input snapshot (frozen at begin) and its outputs. */
let snapshot: OmniMessage[] = [];
let outputs: CompleteModelMessage[] = [];
let inRequest = false;
/** Whether we're between a matched pair of compaction events: the compaction prompt in this
* span is not conversational input and must not be resent as-is if uncommitted. */
let inCompaction = false;
/** Ids of tool_calls that are committed (in history) and ids of outputs that are paired (in
* history input). */
const committedCallIds = new Set<string>();
const answeredIds = new Set<string>();
let sessionTurns = 0;
const renderMessages: OmniMessage[] = [];
const placeholderFor = (id: string): CompleteModelMessage =>
toolCallOutput({
output: PROCESS_EXIT_PLACEHOLDER,
toolCallId: id,
stopReason: "aborted",
}) as CompleteModelMessage;
const dropUncommittedRound = (): void => {
// Uncommitted turn: the whole turn is excluded from history. Its **original input** (user
// text/images and structured tool output) goes back into the pending buffer as-is — best
// effort to resend "the last input that got no response"; incomplete model output
// (thinking/text) is allowed to be lost.
// Exception: a failed compaction turn's compaction prompt is not conversational input and is
// not reclaimed (structured output is still reclaimed, subject to eligibility filtering).
const keep = inCompaction ? snapshot.filter((m) => toolCallOutputId(m) !== null) : snapshot;
pending = [...keep, ...pending];
snapshot = [];
outputs = [];
inRequest = false;
};
for (const msg of messages) {
if (isSessionMeta(msg)) continue;
if (isRequestBegin(msg)) {
// Defensive: the previous turn had begin but no end (shouldn't happen mid-file — a process
// exit only affects the tail) — treat it as uncommitted.
if (inRequest) dropUncommittedRound();
inRequest = true;
snapshot = pending;
pending = [];
continue;
}
if (isRequestEnd(msg)) {
if (msg.payload.status === "completed") {
// Structural eligibility filter (same rule as the final carry-over): a tool_call_output
// in the snapshot is kept only if it pairs with a tool_call that is **committed and not
// yet answered** — tool output dispatched by a dropped turn gets persisted in the next
// turn's input span, but its tool_call isn't in history, so keeping it as-is would create
// an orphan tool_result with no preceding tool_use, which every request would be rejected
// for by the provider after resume ("Replay Rules"' strict pairing guarantee).
const eligible = snapshot.filter((m) => {
const id = toolCallOutputId(m);
return id === null || (committedCallIds.has(id) && !answeredIds.has(id));
});
// Structural repair (best-effort): a tool_call committed earlier but still unpaired — its
// matching output was once sent as synthesized carry-over but **never written to Trace**
// — resynthesize a placeholder before this turn's input, to keep the injected history
// structurally valid (every assistant tool_use is followed by a user tool_result).
const snapshotOutputIds = new Set(
eligible.map(toolCallOutputId).filter((id): id is string => id !== null),
);
for (const id of committedCallIds) {
if (answeredIds.has(id) || snapshotOutputIds.has(id)) continue;
history.push(placeholderFor(id));
answeredIds.add(id);
}
history.push(...(eligible as CompleteModelMessage[]), ...outputs);
for (const id of snapshotOutputIds) answeredIds.add(id);
for (const m of outputs) {
const p = m.payload as Partial<ToolCallPayload>;
if (p.type === "tool_call" && p.stop_reason === "completed" && p.tool_call_id) {
committedCallIds.add(p.tool_call_id);
}
}
sessionTurns += 1;
snapshot = [];
outputs = [];
inRequest = false;
} else {
dropUncommittedRound();
}
continue;
}
if (isCompleteModelMessage(msg)) {
renderMessages.push(msg);
const role = (msg.payload as { role?: string }).role;
if (role === "user") {
// All user-side messages go into the pending buffer: ones between begin/end (output from
// parallel tools completing during the request) count toward the next turn's input.
pending.push(msg);
} else if (inRequest) {
outputs.push(msg);
}
// Defensive: assistant messages outside a span (shouldn't happen) are excluded from
// history, kept only for rendering.
continue;
}
if (isEventMessage(msg)) {
const t = (msg.payload as { type?: string }).type;
if (t === "compaction_begin") inCompaction = true;
else if (t === "compaction_end") inCompaction = false;
else if (t === "abort" && (msg.origin?.length ?? 0) === 0) renderMessages.push(msg);
// Other events (token_usage / approval_decision, etc.) don't participate in turn
// determination.
}
}
// File ends mid-request (begin but no end — the process exited during a request): treat as an
// uncommitted turn.
if (inRequest) dropUncommittedRound();
// —— Structured resend eligibility filter: a tool_call_output in the pending input is kept
// only if it pairs with a tool_call that is **committed and not yet answered**. Tool output
// dispatched by an uncommitted turn itself (which may be persisted before or after that turn's
// end — the tool and LLM streams run concurrently, so finishing order is unpredictable) is
// dropped entirely: its tool_call isn't in history, so a structured resend would form an orphan
// tool_result with no preceding tool_use, which the provider would reject.
pending = pending.filter((m) => {
const id = toolCallOutputId(m);
return id === null || (committedCallIds.has(id) && !answeredIds.has(id));
});
// —— Pairing backfill: for any tool_call committed in history with no matching output in
// either history or pending input, add an interrupted-state placeholder to pending input.
// Placeholders exist only in memory (synthesized carry-over is never written to Trace) and are
// resynthesized on each resume as needed.
const pairedIds = new Set<string>(answeredIds);
for (const m of pending) {
const id = toolCallOutputId(m);
if (id !== null) pairedIds.add(id);
}
const pairingBackfill: OmniMessage[] = [];
for (const id of committedCallIds) {
if (pairedIds.has(id)) continue;
pairingBackfill.push(placeholderFor(id));
}
return {
history,
carryOver: [...pending, ...pairingBackfill],
contextClosed: false,
sessionTokens,
lastRequestTotal,
sessionTurns,
renderMessages,
meta,
};
}
/** The session cumulative total from the last token_usage (zero if none). */
function lastSessionTokens(messages: OmniMessage[]): TokenCounts {
for (let i = messages.length - 1; i >= 0; i--) {
const msg = messages[i]!;
if (!isEventMessage(msg)) continue;
const p = msg.payload as Partial<TokenUsagePayload>;
if (p.type === "token_usage" && p.session) return p.session;
}
return emptyTokenCounts();
}
/** The request.total from the last token_usage (context usage figure; 0 if none). */
function lastRequestTotalOf(messages: OmniMessage[]): number {
for (let i = messages.length - 1; i >= 0; i--) {
const msg = messages[i]!;
if (!isEventMessage(msg)) continue;
const p = msg.payload as Partial<TokenUsagePayload>;
if (p.type === "token_usage" && p.request) return p.request.total;
}
return 0;
}
/** Concatenated assistant text of the last completed Request (on compaction closure, this is the compaction request's output). */
function lastCompletedRequestText(messages: OmniMessage[]): string {
let text = "";
let current = "";
let inRequest = false;
for (const msg of messages) {
if (isRequestBegin(msg)) {
inRequest = true;
current = "";
continue;
}
if (isRequestEnd(msg)) {
{
// A completed request with empty text still overwrites (we take the text of “the last
// completed request”, even if empty) — otherwise a textless compaction output would fall
// back to an earlier turn's normal reply and get mistakenly injected as the summary; the
// in-process path yields an empty summary here (extractSummary(“”)).
if (msg.payload.status === "completed") text = current;
inRequest = false;
}
continue;
}
if (!inRequest || !isCompleteModelMessage(msg)) continue;
const p = msg.payload as { type?: string; role?: string; text?: string };
if (p.type === "text" && p.role === "assistant" && p.text) current += p.text;
}
return text;
}
+149
View File
@@ -0,0 +1,149 @@
/**
* Trace writer — append-only JSON Lines.
*
* Docs: packages/docs/content/sessions-and-traces.{zh,en}.md (site path
* /docs/sessions-and-traces) documents the file layout and recording rules.
*
* Design points:
* - Every observable action is appended to Trace; historical events are never modified in place
* (append-only).
* - One Trace file corresponds to one complete model context; when the context is compacted
* and a new segment is produced, `rotate()` starts a new, separately numbered file.
* - Only "recordable" messages are written: `session_meta`, complete `model_msg`, and all
* `event_msg`; streaming `partial_*` messages are skipped (the producer appends the
* corresponding complete message once the segment ends); nested child-session messages are
* never written (their spawn location is recorded via the `subagent` pointer event written by
* context_engine).
* - Path convention: `<tracesDir>/<yyyy-mm-dd>/<sessionId>_<index3>.jsonl`.
*/
import { appendFile, mkdir, readFile } from "node:fs/promises";
import { dirname, join } from "node:path";
import {
PartialAggregator,
isCompleteModelMessage,
isEventMessage,
isSessionMeta,
} from "../omnimessage/index.js";
import type { OmniMessage } from "../omnimessage/index.js";
import { formatLocalDate } from "../internal/dates.js";
export interface WriterOptions {
/** Trace root directory, typically `<agent>/traces`. */
tracesDir: string;
/** Current Session id, written into the file name. */
sessionId: string;
/** The time used to derive the date subdirectory; defaults to `new Date()`. */
date?: Date;
/**
* Directly specifies the date subdirectory name (used when Session resumption continues
* writing to the original file: the Trace file follows the context, not the date); takes
* priority over `date`.
*/
dateDir?: string;
/** Starting Trace index (used when Session resumption continues the original index); defaults to 1. */
startIndex?: number;
}
/** Zero-pads a Trace index to 3 digits, e.g. 1 -> "001". */
function formatIndex(index: number): string {
return index.toString().padStart(3, "0");
}
/**
* Determines whether an OmniMessage should be written to Trace (skips streaming partial_* and nested child-session messages).
*
* Child-session messages are never written to this Trace: the child Session has its own complete
* Trace, and recording it again would distort this Trace's statistics. The spawn location is
* recorded via the `subagent` pointer event (recording only the child Session id) that
* context_engine writes at the spawn site; when the session is reopened, the server uses this to
* re-attach the child session to its corresponding run_subagent tool card.
* Docs: /docs/sessions-and-traces § "Trace design".
*/
function isRecordable(msg: OmniMessage): boolean {
if (msg.origin && msg.origin.length > 0) return false;
return isCompleteModelMessage(msg) || isEventMessage(msg) || isSessionMeta(msg);
}
/**
* append-only JSONL Trace writer.
*
* Single-writer scenario (MVP): concurrency safety isn't required, but every write uses
* `appendFile` (O_APPEND) rather than caching a file handle and seeking to write, avoiding
* overwriting existing content; this also removes the need for an explicit close.
*/
export class Writer {
private readonly tracesDir: string;
private readonly sessionId: string;
private readonly dateDir: string;
/** Current Trace index, starting at 1; incremented by `rotate()`. */
private index = 1;
/** Set true once the date directory has been created for the current file, to avoid a redundant mkdir. */
private ensuredDirForIndex = -1;
constructor(opts: WriterOptions) {
this.tracesDir = opts.tracesDir;
this.sessionId = opts.sessionId;
this.dateDir = opts.dateDir ?? formatLocalDate(opts.date ?? new Date());
this.index = opts.startIndex ?? 1;
}
/** Absolute path of the current Trace file. */
currentPath(): string {
const fileName = `${this.sessionId}_${formatIndex(this.index)}.jsonl`;
return join(this.tracesDir, this.dateDir, fileName);
}
/**
* Appends one message. Only written if it's a recordable message; streaming `partial_*` is
* skipped. `mkdir -p`s the date directory on the first write to the current file.
*/
async write(msg: OmniMessage): Promise<void> {
if (!isRecordable(msg)) return;
const path = this.currentPath();
if (this.ensuredDirForIndex !== this.index) {
await mkdir(dirname(path), { recursive: true });
this.ensuredDirForIndex = this.index;
}
await appendFile(path, `${JSON.stringify(msg)}\n`, "utf8");
}
/** Writes multiple messages in sequence. */
async writeAll(msgs: OmniMessage[]): Promise<void> {
for (const msg of msgs) {
await this.write(msg);
}
}
/**
* Aggregates a message stream mixed with streaming `partial_*` into complete messages first,
* then writes them per the `write` convention. A convenience helper: `write` skips partial_*
* by default (the producer will already append the complete message), so this method is only
* needed when reconstructing a complete context from raw streaming fragments.
*/
async aggregateAndWrite(msgs: OmniMessage[]): Promise<void> {
const agg = new PartialAggregator();
for (const msg of msgs) {
await this.writeAll(agg.push(msg));
}
await this.writeAll(agg.flush());
}
/**
* Starts a new Trace file: increments the index, so the next `write` goes to the new file.
* Used to split into a separate file when the context is compacted and a new context segment is produced.
* Docs: /docs/sessions-and-traces § "Trace design".
*/
async rotate(): Promise<void> {
this.index += 1;
}
}
/** Parses a Trace file line by line (ignoring blank lines), for testing and later reads. */
export async function readTrace(path: string): Promise<OmniMessage[]> {
const content = await readFile(path, "utf8");
return content
.split("\n")
.filter((line) => line.trim().length > 0)
.map((line) => JSON.parse(line) as OmniMessage);
}
+105
View File
@@ -0,0 +1,105 @@
/**
* Agent lifecycle: new session id format + no more `.penguin` symlink in the Workspace.
*
* - sessionId looks like `session-YYYY-MM-DD-HH-mm-ss-<8-digit hex>` (local time, zero-padded fields).
* - createSession no longer creates any `.penguin` symlink inside the Workspace, nor touches existing
* Workspace files; the model reaches Agent State and other absolute paths directly by combining the
* Project Dir / Agent ID placeholders from the system prompt.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import { createAgent } from "../src/index.js";
import { formatSessionId } from "../src/internal/session-support.js";
import { projectDir } from "../src/state/paths.js";
import { stubProviderKeys } from "./provider-keys.js";
let tmpRoot: string;
let prevHome: string | undefined;
let restoreKeys: () => void;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-harness-lifecycle-"));
process.env.PENGUIN_HOME = tmpRoot;
restoreKeys = stubProviderKeys();
});
afterEach(async () => {
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
else process.env.PENGUIN_HOME = prevHome;
restoreKeys();
await fs.rm(tmpRoot, { recursive: true, force: true });
});
const SESSION_ID_RE = /^session-\d{4}-\d{2}-\d{2}-\d{2}-\d{2}-\d{2}-[0-9a-f]{8}$/;
describe("formatSessionId", () => {
it("matches the session-YYYY-MM-DD-HH-mm-ss-<8hex> format", () => {
expect(formatSessionId()).toMatch(SESSION_ID_RE);
});
it("uses local time fields with zero padding", () => {
// 2026-06-19 15:28:08 local time -> session-2026-06-19-15-28-08-<hex>.
const d = new Date(2026, 5, 19, 15, 28, 8);
expect(formatSessionId(d)).toMatch(/^session-2026-06-19-15-28-08-[0-9a-f]{8}$/);
});
it("generates distinct ids on repeated calls", () => {
expect(formatSessionId()).not.toBe(formatSessionId());
});
});
describe("Agent.createSession session id + no .penguin symlink", () => {
it("assigns a sessionId in the new format", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
expect(session.sessionId).toMatch(SESSION_ID_RE);
});
it("does not create a .penguin entry in the workspace", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws");
await fs.mkdir(ws, { recursive: true });
await agent.createSession({ workspaceDir: ws });
await expect(fs.lstat(path.join(ws, ".penguin"))).rejects.toThrow();
// agent_state also no longer creates traces/notes symlinks.
await expect(fs.lstat(path.join(agent.state.stateDir, "traces"))).rejects.toThrow();
await expect(fs.lstat(path.join(agent.state.stateDir, "notes"))).rejects.toThrow();
});
it("leaves a user's pre-existing .penguin file untouched", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws");
await fs.mkdir(ws, { recursive: true });
const linkPath = path.join(ws, ".penguin");
await fs.writeFile(linkPath, "user-data", "utf8");
await agent.createSession({ workspaceDir: ws });
expect(await fs.readFile(linkPath, "utf8")).toBe("user-data");
});
it("is idempotent: repeated createSession registers no exit listeners", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws");
await fs.mkdir(ws, { recursive: true });
const before = process.listenerCount("exit");
for (let i = 0; i < 12; i++) {
await agent.createSession({ workspaceDir: ws });
}
expect(process.listenerCount("exit") - before).toBe(0);
});
it("injects Project Dir and Agent ID into the assembled system prompt", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
const prompt = (session.metaMessage.payload as { system_prompt: string }).system_prompt;
expect(prompt).toContain(`Agent ID: ${agent.state.agentId}`);
expect(prompt).toContain(`Project Dir: ${projectDir(tmpRoot, agent.state.projectId)}`);
expect(prompt).not.toContain(".penguin");
});
});
+201
View File
@@ -0,0 +1,201 @@
/**
* On-disk behavior of an Agent's installed Skills: installSkill /
* removeSkill / listInstalledSkills, and metadata injection via skillMetadataSection /
* assembleSystemPrompt.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import { librarySkill } from "@prismshadow/penguin-skills";
import {
AGENTS_MD_PLACEHOLDER,
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
SKILL_METADATA_PLACEHOLDER,
agentStateDir,
assembleSystemPrompt,
installSkill,
listInstalledSkills,
removeSkill,
skillMetadataSection,
skillsDir,
} from "../src/state/index.js";
let tmpRoot: string;
beforeEach(async () => {
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-skills-"));
});
afterEach(async () => {
await fs.rm(tmpRoot, { recursive: true, force: true });
});
const install = (name: string, content: string, icon?: string) =>
installSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, {
name,
content,
...(icon !== undefined ? { icon } : {}),
});
const list = () => listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID);
const skillFile = (name: string, file: string) =>
path.join(skillsDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), name, file);
const skillMd = (name: string) => skillFile(name, "SKILL.md");
const skillIcon = (name: string) => skillFile(name, "icon.svg");
describe("installSkill / removeSkill", () => {
it("writes skills/<name>/SKILL.md verbatim with a trailing newline", async () => {
const skill = librarySkill("penguin-cli")!;
await install(skill.name, skill.content);
expect(await fs.readFile(skillMd("penguin-cli"), "utf8")).toBe(skill.content);
// Content without a trailing newline gets one appended; reinstalling overwrites.
await install("penguin-cli", "---\nname: penguin-cli\nversion: 2\n---\n\nNew body");
expect(await fs.readFile(skillMd("penguin-cli"), "utf8")).toBe(
"---\nname: penguin-cli\nversion: 2\n---\n\nNew body\n",
);
expect((await list()).map((s) => s.version)).toEqual([2]);
});
it("writes icon.svg alongside SKILL.md, and reinstalling without icon removes it", async () => {
// A library skill with an icon: installing writes it to disk alongside SKILL.md.
const skill = librarySkill("penguin-sdk")!;
expect(skill.icon).toBeTruthy();
await install(skill.name, skill.content, skill.icon);
expect(await fs.readFile(skillIcon("penguin-sdk"), "utf8")).toBe(skill.icon);
// Overwrite semantics: this install has no icon -> the old icon.svg is removed, and the
// directory matches this install's content exactly.
await install("penguin-sdk", "---\nname: penguin-sdk\nversion: 2\n---\n\nNew body\n");
await expect(fs.access(skillIcon("penguin-sdk"))).rejects.toThrow();
expect(await fs.readFile(skillMd("penguin-sdk"), "utf8")).toContain("New body");
});
it("rejects invalid skill names (path traversal safety)", async () => {
await expect(install("../evil", "x")).rejects.toThrow(/skill_name/);
await expect(install("a/b", "x")).rejects.toThrow(/skill_name/);
await expect(removeSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "..")).rejects.toThrow(
/skill_name/,
);
});
it("removeSkill deletes the whole skill directory and is idempotent", async () => {
const skill = librarySkill("penguin-sdk")!;
await install(skill.name, skill.content);
await removeSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "penguin-sdk");
await expect(fs.access(skillMd("penguin-sdk"))).rejects.toThrow();
// Idempotent when it no longer exists: does not throw.
await removeSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, "penguin-sdk");
expect(await list()).toEqual([]);
});
});
describe("listInstalledSkills", () => {
it("returns [] when the skills directory does not exist", async () => {
expect(await list()).toEqual([]);
});
it("parses frontmatter and sorts by name", async () => {
await install(
"zeta",
"---\nname: zeta\ndescription: Z skill.\nversion: 3\nupdated: 2026-07-16\n---\n\nBody\n",
);
await install(
"alpha",
"---\nname: alpha\ndescription: A skill.\nversion: 1\nupdated: 2026-07-16\n---\n\nBody\n",
);
expect(await list()).toEqual([
{ name: "alpha", description: "A skill.", version: 1, updated: "2026-07-16" },
{ name: "zeta", description: "Z skill.", version: 3, updated: "2026-07-16" },
]);
});
it("returns icon.svg content and passes short description fields through", async () => {
const icon = '<svg viewBox="0 0 24 24"><path d="M4 4h16" /></svg>\n';
await install(
"with-extras",
"---\nname: with-extras\ndescription: Long description here.\nshort_description: Short one.\nshort_description_zh: 短描述。\nversion: 1\nupdated: 2026-07-17\n---\n\nBody\n",
icon,
);
await install(
"plain",
"---\nname: plain\ndescription: Plain skill.\nversion: 1\nupdated: 2026-07-17\n---\n\nBody\n",
);
const skills = await list();
expect(skills).toEqual([
{ name: "plain", description: "Plain skill.", version: 1, updated: "2026-07-17" },
{
name: "with-extras",
description: "Long description here.",
shortDescription: "Short one.",
shortDescriptionZh: "短描述。",
version: 1,
updated: "2026-07-17",
icon,
},
]);
// Entries missing icon / short description omit the corresponding field (undefined does
// not produce a key; the interface layer's conditional spread relies on this convention).
expect("icon" in skills[0]!).toBe(false);
expect("shortDescription" in skills[0]!).toBe(false);
});
it("uses the directory name as the skill identity even when frontmatter name disagrees", async () => {
// A hand-written or network-sourced skill may have a frontmatter name that differs from
// its directory name: the directory name is the addressing key used for install, uninstall,
// and Prompt lookup, so the listing must follow it (frontmatter fields are display-only).
await install(
"local-name",
"---\nname: upstream-name\ndescription: Fetched skill.\nversion: 2\nupdated: 2026-07-01\n---\n\nBody\n",
);
expect(await list()).toEqual([
{ name: "local-name", description: "Fetched skill.", version: 2, updated: "2026-07-01" },
]);
});
it("falls back to directory-name metadata for broken frontmatter and skips non-skill entries", async () => {
// A SKILL.md without frontmatter: falls back to directory name + empty description + version 1.
await install("broken", "# No frontmatter here\n");
// Directories without a SKILL.md, and stray files, do not count as Skills.
const dir = skillsDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID);
await fs.mkdir(path.join(dir, "empty-dir"), { recursive: true });
await fs.writeFile(path.join(dir, "stray.md"), "stray", "utf8");
expect(await list()).toEqual([{ name: "broken", description: "", version: 1, updated: "" }]);
});
});
describe("skillMetadataSection / assembleSystemPrompt 注入", () => {
it("renders one `- \\`name\\` — description` line per skill; empty input renders empty", () => {
expect(skillMetadataSection([])).toBe("");
expect(
skillMetadataSection([
{ name: "a", description: "Does A.", version: 1, updated: "2026-07-16" },
{ name: "b", description: "", version: 1, updated: "" },
]),
).toBe("- `a` — Does A.\n- `b`");
});
it("replaces {{SKILL_METADATA}} with metadata lines, or an empty string when absent", () => {
const state = {
root: tmpRoot,
projectId: DEFAULT_PROJECT_ID,
agentId: DEFAULT_AGENT_ID,
stateDir: agentStateDir(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID),
systemConfig: {
system_prompt: ["before", AGENTS_MD_PLACEHOLDER, SKILL_METADATA_PLACEHOLDER, "after"].join(
"\n",
),
},
agentsMd: "# Agent Rules",
};
const prompt = assembleSystemPrompt(state, undefined, undefined, [
{ name: "demo", description: "Demo skill.", version: 1, updated: "2026-07-16" },
]);
expect(prompt).toBe(["before", "# Agent Rules", "- `demo` — Demo skill.", "after"].join("\n"));
// Not provided / empty list: the placeholder is replaced with an empty string, no residue left.
const empty = assembleSystemPrompt(state);
expect(empty).toBe(["before", "# Agent Rules", "", "after"].join("\n"));
expect(empty).not.toContain(SKILL_METADATA_PLACEHOLDER);
});
});
+256
View File
@@ -0,0 +1,256 @@
/**
* Agent.createSession's Workspace handling and vault injection (no network needed; only
* constructs the Session, never sends a request).
*
* Regression: an explicitly given Workspace must be an existing directory. When it
* does not exist, a clear error must be thrown rather than auto-creating it, and bash must not
* be started with an invalid cwd after Session creation, which would throw a misleading
* `spawn bash ENOENT`. A temp directory is only created when no Workspace is specified.
*
* vault: the Agent vault's (agent_state/.vault.toml) **key names** are
* injected into the assembled system prompt; values are never injected.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import {
addModel,
createAgent,
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
installSkill,
setVaultEntry,
} from "../src/index.js";
import { effectiveMaxContextLength } from "../src/agent.js";
import { stubProviderKeys } from "./provider-keys.js";
let tmpRoot: string;
let prevHome: string | undefined;
let restoreKeys: () => void;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-harness-"));
process.env.PENGUIN_HOME = tmpRoot;
restoreKeys = stubProviderKeys();
});
afterEach(async () => {
if (prevHome === undefined) delete process.env.PENGUIN_HOME;
else process.env.PENGUIN_HOME = prevHome;
restoreKeys();
await fs.rm(tmpRoot, { recursive: true, force: true });
});
describe("effectiveMaxContextLength (压缩阈值按模型窗口钳制)", () => {
it("clamps to 75% of a small model window; leaves big/unknown windows and off untouched", () => {
expect(effectiveMaxContextLength(128000, 32768)).toBe(24576); // small window: clamp to 75%
expect(effectiveMaxContextLength(128000, 200000)).toBe(128000); // ample window: unchanged
expect(effectiveMaxContextLength(-1, 32768)).toBe(-1); // off: no clamping
expect(effectiveMaxContextLength(0, 32768)).toBe(0); // off: no clamping
expect(effectiveMaxContextLength(128000, "unknown")).toBe(128000); // unknown window: no clamping
});
});
describe("Agent.createSession workspace handling", () => {
it("throws a clear error when the given workspace does not exist (no auto-create)", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "nested", "does-not-exist");
await expect(agent.createSession({ workspaceDir: ws })).rejects.toThrow(/不存在/);
// Must not be auto-created.
await expect(fs.stat(ws)).rejects.toThrow();
});
it("throws when the given workspace path is not a directory", async () => {
const agent = await createAgent();
const filePath = path.join(tmpRoot, "a-file");
await fs.writeFile(filePath, "x", "utf8");
await expect(agent.createSession({ workspaceDir: filePath })).rejects.toThrow(/不是目录/);
});
it("accepts an existing directory and resolves it to an absolute path", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
expect(session.workspaceDir).toBe(ws);
expect(path.isAbsolute(session.workspaceDir)).toBe(true);
});
it("rejects a modelId that is not in the Project config with a clear error", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-bad-model");
await fs.mkdir(ws, { recursive: true });
// A reference outside the config is not silently allowed (the unique key is provider +
// model_id); the error is thrown before creating the temp Workspace.
await expect(
agent.createSession({ workspaceDir: ws, modelId: "not-configured-model" }),
).rejects.toThrow(/不在 Project 配置中/);
await expect(
agent.createSession({ workspaceDir: ws, modelId: "deepseek-v4-pro", provider: "openai" }),
).rejects.toThrow(/\(provider=openai, model_id=deepseek-v4-pro\)/);
});
it("passes model timeout from system_config to GenerativeModel", async () => {
const agent = await createAgent();
agent.state.systemConfig.model = {
...(agent.state.systemConfig.model ?? {}),
timeoutMs: 3456,
};
const ws = path.join(tmpRoot, "ws-timeout");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
const llm = (session as unknown as { engine: { deps: { llm: unknown } } }).engine.deps.llm;
expect((llm as { requestTimeoutMs?: number }).requestTimeoutMs).toBe(3456);
});
});
describe("Agent.createSession model reference((provider, model_id) 成对)", () => {
it("records the pair reference in session_meta (default_model when unspecified)", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-ref-default");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
try {
// Defaults to the default_model reference; session_meta carries the pair reference
// (same source that Trace writes).
const meta = session.metaMessage.payload as { provider: string; model_id: string };
expect(meta.provider).toBe("deepseek");
expect(meta.model_id).toBe("deepseek-v4-pro");
expect(session.provider).toBe("deepseek");
expect(session.modelId).toBe("deepseek-v4-pro");
} finally {
session.dispose();
}
});
it("resolves a unique bare model_id and accepts an explicit pair", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-ref-pair");
await fs.mkdir(ws, { recursive: true });
// Provider omitted: model_id is a globally unique exact match in the config -> resolves to that entry.
const bare = await agent.createSession({ workspaceDir: ws, modelId: "deepseek-v4-flash" });
try {
expect(bare.provider).toBe("deepseek");
expect(bare.modelId).toBe("deepseek-v4-flash");
} finally {
bare.dispose();
}
const paired = await agent.createSession({
workspaceDir: ws,
modelId: "claude-sonnet-4-6",
provider: "anthropic",
});
try {
expect(paired.provider).toBe("anthropic");
expect(paired.modelId).toBe("claude-sonnet-4-6");
} finally {
paired.dispose();
}
});
it("rejects an ambiguous bare model_id and a provider without modelId", async () => {
// Two providers coexist with the same model_id: omitting provider throws an ambiguity
// error (listing the candidate pair references).
await addModel(tmpRoot, DEFAULT_PROJECT_ID, {
provider: "myproxy",
model_id: "claude-sonnet-4-6",
});
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-ref-ambiguous");
await fs.mkdir(ws, { recursive: true });
await expect(
agent.createSession({ workspaceDir: ws, modelId: "claude-sonnet-4-6" }),
).rejects.toThrow(/歧义.*\(provider=anthropic, model_id=claude-sonnet-4-6\)/);
// Adding provider resolves it.
const session = await agent.createSession({
workspaceDir: ws,
modelId: "claude-sonnet-4-6",
provider: "myproxy",
});
try {
expect(session.provider).toBe("myproxy");
} finally {
session.dispose();
}
// provider cannot be used alone (the reference must be a pair).
await expect(agent.createSession({ workspaceDir: ws, provider: "anthropic" })).rejects.toThrow(
/provider 不能单独使用/,
);
});
});
describe("Agent.createSession vault injection", () => {
it("injects vault key names (never values) into the assembled system prompt", async () => {
// Write the Agent vault to disk first; createSession reads that Agent's own .vault.toml.
await setVaultEntry(
tmpRoot,
DEFAULT_PROJECT_ID,
DEFAULT_AGENT_ID,
"VAULT_ONLY_KEY",
"vault-secret-value",
);
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-vault");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
try {
const meta = session.metaMessage.payload as { system_prompt: string };
// The "# Vault" statement is part of the template body; key names are injected at the
// placeholder as a `- KEY` list.
expect(meta.system_prompt).toContain("# Vault");
expect(meta.system_prompt).toContain("- VAULT_ONLY_KEY");
// Values never enter the model context.
expect(meta.system_prompt).not.toContain("vault-secret-value");
} finally {
session.dispose();
}
});
it("keeps the vault statement but lists no keys when the Agent has no vault", async () => {
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-no-vault");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
try {
const meta = session.metaMessage.payload as { system_prompt: string };
// No vault: the "# Vault" section statement is kept, and the
// placeholder is replaced with an empty string, leaving no residue.
expect(meta.system_prompt).toContain("# Vault");
expect(meta.system_prompt).not.toContain("{{VAULT_KEYS}}");
expect(meta.system_prompt).not.toContain("VAULT_ONLY_KEY");
} finally {
session.dispose();
}
});
});
describe("Agent.createSession skill metadata injection", () => {
it("injects installed skill metadata lines (never bodies) into the assembled system prompt", async () => {
await installSkill(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID, {
name: "demo-skill",
content:
"---\nname: demo-skill\ndescription: Demo skill for tests.\nversion: 1\nupdated: 2026-07-16\n---\n\nSKILL_BODY_NOT_IN_PROMPT\n",
});
const agent = await createAgent();
const ws = path.join(tmpRoot, "ws-skills");
await fs.mkdir(ws, { recursive: true });
const session = await agent.createSession({ workspaceDir: ws });
try {
const meta = session.metaMessage.payload as { system_prompt: string };
expect(meta.system_prompt).toContain("# Skills");
expect(meta.system_prompt).toContain("- `demo-skill` — Demo skill for tests.");
// Only metadata is injected; the model reads the body on demand.
expect(meta.system_prompt).not.toContain("SKILL_BODY_NOT_IN_PROMPT");
expect(meta.system_prompt).not.toContain("{{SKILL_METADATA}}");
} finally {
session.dispose();
}
});
});
@@ -0,0 +1,64 @@
/**
* Behavior tests for BackgroundRegistry's idle reaping (a leak safety net).
*/
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { BackgroundRegistry } from "../src/environment/tools/background/index.js";
import type { BackgroundTask } from "../src/environment/tools/background/index.js";
type FakeTask = BackgroundTask & { killed: boolean };
function fakeTask(): FakeTask {
const task: FakeTask = {
lastUsed: 0,
running: true,
killed: false,
kill() {
task.killed = true;
},
killHard() {
task.killed = true;
},
};
return task;
}
describe("BackgroundRegistry idle reaping", () => {
beforeEach(() => {
vi.useFakeTimers();
});
afterEach(() => {
vi.useRealTimers();
});
it("reaps sessions idle past the TTL and keeps recently accessed ones", () => {
const registry = new BackgroundRegistry<FakeTask>({ idPrefix: "proc", maxTasks: 4 });
const stale = fakeTask();
const fresh = fakeTask();
const staleId = registry.register(stale);
const freshId = registry.register(fresh);
// After 9 days, one access to fresh refreshes its lastUsed; stale is never accessed.
vi.advanceTimersByTime(9 * 24 * 60 * 60_000);
expect(registry.get(freshId)).toBe(fresh);
// stale, now idle a full 10 days, is reaped by the scheduled sweep and finalized;
// fresh has been idle only 1 day and is kept.
vi.advanceTimersByTime(24 * 60 * 60_000 + 60 * 60_000);
expect(registry.get(staleId)).toBeUndefined();
expect(stale.killed).toBe(true);
expect(registry.get(freshId)).toBe(fresh);
expect(fresh.killed).toBe(false);
registry.dispose();
});
it("stops the reap timer on dispose", () => {
const registry = new BackgroundRegistry<FakeTask>({ idPrefix: "proc", maxTasks: 4 });
const task = fakeTask();
registry.register(task);
registry.dispose();
expect(task.killed).toBe(true);
// After dispose the sweep timer is cleared, so fast-forwarding no longer triggers any reaping logic.
expect(() => vi.advanceTimersByTime(30 * 24 * 60 * 60_000)).not.toThrow();
});
});
+141
View File
@@ -0,0 +1,141 @@
/**
* Built-in agent provisioning and skill library install policy: the sole built-in agent
* default_agent comes pre-installed with every skill in the library, an ordinary newly created
* agent starts with zero skills, and the default AGENTS.md is an empty file; provisionProjectAgents
* is idempotent and never overwrites existing config.
*/
import fs from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import { librarySkill, loadLibrarySkills } from "@prismshadow/penguin-skills";
import {
agentsMdPath,
assembleSystemPrompt,
BUILTIN_AGENT_IDS,
DEFAULT_AGENT_ID,
DEFAULT_PROJECT_ID,
listInstalledSkills,
loadOrInitAgentState,
provisionProjectAgents,
skillsDir,
} from "../src/state/index.js";
let tmpRoot: string;
let prevHome: string | undefined;
beforeEach(async () => {
prevHome = process.env.PENGUIN_HOME;
tmpRoot = await fs.mkdtemp(path.join(os.tmpdir(), "penguin-builtin-"));
process.env.PENGUIN_HOME = tmpRoot;
});
afterEach(async () => {
if (prevHome === undefined) {
delete process.env.PENGUIN_HOME;
} else {
process.env.PENGUIN_HOME = prevHome;
}
await fs.rm(tmpRoot, { recursive: true, force: true });
});
const skillMdPath = (agentId: string, skillName: string): string =>
path.join(skillsDir(tmpRoot, DEFAULT_PROJECT_ID, agentId), skillName, "SKILL.md");
describe("Skill 安装策略", () => {
it("普通新建 Agent 不预装任何 Skill,AGENTS.md 为空文件(指导在模板 Suggested workflows)", async () => {
const state = await loadOrInitAgentState({ agentId: "some_agent" });
expect(await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, "some_agent")).toEqual([]);
// The default AGENTS.md is empty: it carries no preset guidance (delegation and task
// conventions live in the default template's Suggested workflows section, and skill
// metadata is injected via {{SKILL_METADATA}}).
expect(state.agentsMd).toBe("");
const onDisk = await fs.readFile(
agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, "some_agent"),
"utf8",
);
expect(onDisk).toBe("");
});
it("无 preset 直建的 default_agent(如 CLI 首次运行)同样预装库内全部 Skill", async () => {
await loadOrInitAgentState({ agentId: DEFAULT_AGENT_ID });
const names = (await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID)).map(
(s) => s.name,
);
expect(names).toEqual(loadLibrarySkills().map((s) => s.name));
});
});
describe("provisionProjectAgents", () => {
it("唯一内置 Agent default_agent:装库内全部 Skill、AGENTS.md 为空", async () => {
const ids = await provisionProjectAgents();
expect(ids).toEqual([DEFAULT_AGENT_ID]);
expect(BUILTIN_AGENT_IDS).toEqual([DEFAULT_AGENT_ID]);
// name/description are written into system_config; AGENTS.md is an empty file.
const state = await loadOrInitAgentState({ agentId: DEFAULT_AGENT_ID });
expect(state.systemConfig.name).toBe("General Agent");
expect(state.systemConfig.description).toBeTruthy();
expect(state.agentsMd).toBe("");
const installed = await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID);
expect(installed.map((s) => s.name).sort()).toEqual(loadLibrarySkills().map((s) => s.name));
// On-disk content matches the library's SKILL.md verbatim (install copies the full text).
const sdkMd = await fs.readFile(skillMdPath(DEFAULT_AGENT_ID, "penguin-sdk"), "utf8");
expect(sdkMd).toBe(librarySkill("penguin-sdk")!.content);
});
it("provision 幂等:重复执行不改变结果", async () => {
await provisionProjectAgents();
const ids = await provisionProjectAgents();
expect(ids).toEqual([DEFAULT_AGENT_ID]);
const md = await fs.readFile(
agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID),
"utf8",
);
expect(md).toBe("");
const installed = await listInstalledSkills(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID);
expect(installed.map((s) => s.name).sort()).toEqual(loadLibrarySkills().map((s) => s.name));
});
it("已存在的 Agent 不被覆盖(preset 仅初始化生效)", async () => {
const custom = "# AGENTS.md\n\n用户自己改过的内容\n";
await loadOrInitAgentState({ agentId: DEFAULT_AGENT_ID });
await fs.writeFile(agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID), custom, "utf8");
await provisionProjectAgents();
const after = await fs.readFile(
agentsMdPath(tmpRoot, DEFAULT_PROJECT_ID, DEFAULT_AGENT_ID),
"utf8",
);
expect(after).toBe(custom);
});
});
describe("Project Dir / Agent ID 占位符", () => {
it("assembleSystemPrompt 注入 Project Dir 与 Agent ID(Skill 定位改走项目相对路径,不依赖 .penguin)", async () => {
const state = await loadOrInitAgentState({ agentId: "env_agent" });
const prompt = assembleSystemPrompt(state, {
sessionId: "session-x",
cwd: "/tmp/ws",
agentId: "env_agent",
projectDir: "/tmp/proj",
platform: "linux",
osVersion: "test",
date: "2026-07-08",
});
expect(prompt).toContain("Agent ID: env_agent");
expect(prompt).toContain("Project Dir: /tmp/proj");
expect(prompt).not.toContain("{{AGENT_ID}}");
expect(prompt).not.toContain("{{PROJECT_DIR}}");
// Skill execution conventions are built from project-relative paths (no longer reference .penguin).
expect(prompt).not.toContain(".penguin");
// Agent State, scratchpad and Skills are addressed under the Project's agents/ container.
expect(prompt).toContain("<project_dir>/agents/<agent_id>/agent_state/");
expect(prompt).toContain("<project_dir>/agents/<agent_id>/agent_state/skills/");
// The pre-agents/ layout (an agent directly under the Project dir) must never be handed to the model.
expect(prompt).not.toContain("<project_dir>/<agent_id>/");
});
});
+683
View File
@@ -0,0 +1,683 @@
/**
* Context compaction tests.
*
* - Trigger: context usage (the request.total of the most recent token_usage) or the session's
* cumulative turn count **reaching** the threshold (>=); the check runs after every LLM
* request emits token_usage, both mid-task and at the wrap-up round (reaching the threshold at
* task end triggers compaction immediately, without waiting for the next task).
* - summarize: appends a compaction prompt to the old LLM (merging in all of this round's tool
* results first if mid-task); the summary is wrapped as a `<context_summary>` user text and fed
* as the first input to the new LLM instance; on failure the original context is kept, never downgraded to discard.
* - discard: deferred until task end if mid-task; sends no compaction request, just swaps in a new LLM instance directly.
* - Process visibility: the compaction request's streamed output is never surfaced to the human,
* only the paired compaction events are emitted; the dialogue is written to the old trace, and
* on success the trace rotates into a new file (index+1, the new file starts with session_meta;
* rotation is deferred until the new context has its first message to write).
*/
import { mkdtemp, rm, access, readdir } from "node:fs/promises";
import { tmpdir } from "node:os";
import { dirname, join } from "node:path";
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import {
assistantText,
sessionMeta,
tokenUsage,
toolCall,
toolCallOutput,
userText,
} from "../src/omnimessage/index.js";
import type {
CompactionBeginPayload,
CompactionEndPayload,
OmniMessage,
TextPayload,
TokenCounts,
TokenUsagePayload,
} from "../src/omnimessage/index.js";
import type {
ApproveFn,
EnvironmentInterface,
GenerativeModelParameters,
LLMInterface,
LLMOutcome,
} from "../src/interfaces.js";
import { ContextEngine } from "../src/engine/context-engine.js";
import type { CompactionSettings } from "../src/engine/context-engine.js";
import { Writer, readTrace } from "../src/trace/index.js";
// ---------------------------------------------------------------------------
// Test fixtures
// ---------------------------------------------------------------------------
interface ScriptedResponse {
messages: OmniMessage[];
outcome?: LLMOutcome;
}
/** Fake LLM that responds according to a script, recording each input it receives. */
class ScriptedLLM implements LLMInterface {
calls: OmniMessage[][] = [];
constructor(
private readonly responses: ScriptedResponse[],
readonly label = "llm",
) {}
async *streamGenerate(
params: GenerativeModelParameters,
): AsyncGenerator<OmniMessage, LLMOutcome> {
this.calls.push(params.newMessages);
const next = this.responses.shift();
if (!next) {
return { status: "failed", message: `${this.label}: no scripted response` };
}
for (const msg of next.messages) yield msg;
return next.outcome ?? { status: "completed" };
}
}
/** Fake Environment that never runs real commands: any tool call returns a fixed output. */
const fakeEnvironment: EnvironmentInterface = {
async listTools() {
return [];
},
async *executeTool({ toolCall: tc }) {
yield toolCallOutput({
output: "tool ran",
toolCallId: tc.payload.tool_call_id,
});
},
toolPermission() {
return "rw";
},
};
const allowAll: ApproveFn = async () => "allow";
/** Builds a token_usage: request.total is the context-usage figure, session.total is the cumulative one. */
const usage = (requestTotal: number, sessionTotal: number): OmniMessage =>
tokenUsage(
{ cache_read: 0, cache_write: 0, output: 0, total: sessionTotal },
{ cache_read: 0, cache_write: 0, output: 0, total: requestTotal },
);
const settings = (over: Partial<CompactionSettings> = {}): CompactionSettings => ({
maxContextLength: 100,
maxSessionTurns: -1,
mode: "summarize",
prompt: "COMPACT NOW",
...over,
});
const metaMessage = sessionMeta({
session_id: "sess_compact",
provider: "custom",
model_id: "test-model",
model_context_window: 200000,
system_prompt: "sp",
tools: [],
thinking_level: "default",
agent_state: "/tmp/state",
workspace: "/tmp/ws",
});
async function collect(gen: AsyncGenerator<OmniMessage>): Promise<OmniMessage[]> {
const all: OmniMessage[] = [];
for await (const msg of gen) all.push(msg);
return all;
}
type CompactionEventPayload = CompactionBeginPayload | CompactionEndPayload;
const compactionEvents = (msgs: OmniMessage[]): CompactionEventPayload[] =>
msgs
.filter((m) => {
const t = (m.payload as { type?: string }).type ?? "";
return t === "compaction_begin" || t === "compaction_end";
})
.map((m) => m.payload as CompactionEventPayload);
const payloadTypes = (msgs: OmniMessage[]): (string | undefined)[] =>
msgs.map((m) => (m.payload as { type?: string }).type);
const textOf = (m: OmniMessage): string => (m.payload as TextPayload).text;
// ---------------------------------------------------------------------------
// Tests
// ---------------------------------------------------------------------------
describe("context compaction", () => {
let traces: string;
beforeEach(async () => {
traces = await mkdtemp(join(tmpdir(), "penguin-compaction-"));
});
afterEach(async () => {
await rm(traces, { recursive: true, force: true });
});
it("summarize at task boundary: paired events, hidden dialogue, trace rotation, summary joins next prompt", async () => {
const llm1 = new ScriptedLLM(
[
// Task 1's final reply: context usage 150 > threshold 100 -> triggers at the boundary.
{ messages: [assistantText("answer one"), usage(150, 150)] },
// Compaction request: summary + usage (counted into the session cumulative total).
{
messages: [assistantText("<summary>the distilled summary</summary>"), usage(160, 310)],
},
],
"llm1",
);
const llm2 = new ScriptedLLM(
[{ messages: [assistantText("answer two"), usage(20, 330)] }],
"llm2",
);
let factoryTokens: TokenCounts | null = null;
const trace = new Writer({ tracesDir: traces, sessionId: "sess_compact" });
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
trace,
sessionMeta: metaMessage,
compaction: settings(),
createLLM: (tokens) => {
factoryTokens = tokens;
return llm2;
},
});
const oldPath = trace.currentPath();
const out1 = await collect(engine.run([userText("task one")], { approve: allowAll }));
// Paired compaction events: start carries reason/mode/context/turns, stop carries status.
const events = compactionEvents(out1);
expect(events).toHaveLength(2);
expect(events[0]).toMatchObject({
type: "compaction_begin",
reason: "context",
mode: "summarize",
context: 150,
turns: 1,
});
expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" });
// The compaction process is invisible to the human: the compaction prompt and summary text are never pushed to the output stream.
const texts = out1.filter((m) => (m.payload as { type?: string }).type === "text").map(textOf);
expect(texts.some((t) => t.includes("COMPACT NOW"))).toBe(false);
expect(texts.some((t) => t.includes("distilled"))).toBe(false);
// Exception: the compaction request's token_usage IS pushed to the output stream, sitting between the paired compaction events (the frontend counts it in stats).
const types1 = payloadTypes(out1);
const between = out1.slice(
types1.indexOf("compaction_begin") + 1,
types1.lastIndexOf("compaction_end"),
);
const usageBetween = between.filter(
(m) => (m.payload as { type?: string }).type === "token_usage",
);
expect(usageBetween).toHaveLength(1);
expect((usageBetween[0]!.payload as TokenUsagePayload).request.total).toBe(160);
// The new LLM instance carries over the session's cumulative tokens (including compaction request usage).
expect(factoryTokens).toMatchObject({ total: 310 });
// The summary is merged with the next user prompt as the new LLM instance's first input.
await collect(engine.run([userText("task two")], { approve: allowAll }));
expect(llm1.calls).toHaveLength(2);
expect(llm2.calls).toHaveLength(1);
const firstInput = llm2.calls[0]!.map(textOf);
expect(firstInput[0]).toBe("<context_summary>\nthe distilled summary\n</context_summary>");
expect(firstInput[1]).toBe("task two");
// Trace splits into files: the old file contains the compaction dialogue and paired events; the new file starts with session_meta.
const oldTrace = await readTrace(oldPath);
const oldTypes = payloadTypes(oldTrace);
expect(oldTypes.filter((t) => t?.startsWith("compaction_"))).toHaveLength(2);
expect(oldTrace.some((m) => (m.payload as { text?: string }).text === "COMPACT NOW")).toBe(
true,
);
const newTrace = await readTrace(trace.currentPath());
expect(trace.currentPath()).not.toBe(oldPath);
expect(newTrace[0]!.type).toBe("session_meta");
expect(
newTrace.some((m) =>
((m.payload as { text?: string }).text ?? "").startsWith("<context_summary>"),
),
).toBe(true);
});
it("summarize mid-task: tool outputs pair into the compaction request, summary alone feeds the new LLM", async () => {
const llm1 = new ScriptedLLM(
[
// Round 1: tool call + over-threshold usage -> triggers mid-task.
{
messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)],
},
// Compaction request (should include c1's tool_call_output plus the compaction prompt).
{ messages: [assistantText("<summary>continue: finish step 2</summary>")] },
],
"llm1",
);
const llm2 = new ScriptedLLM(
[{ messages: [assistantText("task done"), usage(30, 200)] }],
"llm2",
);
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
sessionMeta: metaMessage,
compaction: settings(),
createLLM: () => llm2,
});
const out = await collect(engine.run([userText("do task")], { approve: allowAll }));
// Compaction request: all of this round's tool results, paired with their tool_calls, are sent to the old instance along with the compaction prompt.
expect(llm1.calls).toHaveLength(2);
const compactionInput = llm1.calls[1]!;
const inputTypes = payloadTypes(compactionInput);
expect(inputTypes).toEqual(["tool_call_output", "text"]);
expect((compactionInput[1]!.payload as TextPayload).text).toBe("COMPACT NOW");
// The summary itself is the new instance's first input (no hardcoded continuation instruction appended); the task is finished by the new context.
expect(llm2.calls).toHaveLength(1);
expect(llm2.calls[0]!.map(textOf)).toEqual([
"<context_summary>\ncontinue: finish step 2\n</context_summary>",
]);
const finalTexts = out
.filter((m) => (m.payload as { type?: string }).type === "text")
.map(textOf);
expect(finalTexts).toContain("task done");
expect(compactionEvents(out).map((e) => e.type)).toEqual([
"compaction_begin",
"compaction_end",
]);
});
it("summarize failure keeps the old context and does NOT downgrade to discard", async () => {
const llm1 = new ScriptedLLM(
[
{
messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)],
},
// Compaction request fails (not retryable).
{ messages: [], outcome: { status: "failed", message: "auth error" } },
// Original context is kept: the task continues, tool outputs feed back into the old instance as usual (context usage keeps growing).
{ messages: [assistantText("finished on old context"), usage(190, 340)] },
// Second trigger (context still over the limit) -> retries compaction at the boundary, this time succeeding.
{ messages: [assistantText("<summary>second try</summary>")] },
],
"llm1",
);
const llm2 = new ScriptedLLM([], "llm2");
let created = 0;
const trace = new Writer({ tracesDir: traces, sessionId: "sess_keep" });
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
trace,
sessionMeta: metaMessage,
compaction: settings(),
createLLM: () => {
created += 1;
return llm2;
},
});
const oldPath = trace.currentPath();
const out = await collect(engine.run([userText("go")], { approve: allowAll }));
// Two event pairs: the first has stop=failed (abandoned, original context kept), the second succeeds.
const events = compactionEvents(out);
expect(
events.map((e) => `${e.type}:${(e as Partial<CompactionEndPayload>).status ?? ""}`),
).toEqual([
"compaction_begin:",
"compaction_end:failed",
"compaction_begin:",
"compaction_end:completed",
]);
// No LLM swap and no trace file split at the moment of failure; rotation happens only after success.
expect(created).toBe(1);
expect(llm1.calls).toHaveLength(4);
// After the failure, the input fed back into the old instance is this round's tool output.
expect(payloadTypes(llm1.calls[2]!)).toEqual(["tool_call_output"]);
const oldTrace = await readTrace(oldPath);
// Still written to the same file after a failed stop (the failed compaction attempt stays auditable); the old file is closed off only after success.
expect(payloadTypes(oldTrace).filter((t) => t?.startsWith("compaction_"))).toHaveLength(4);
// Trace rotation is deferred until the next message to write: the current path is unchanged right after a successful compaction.
expect(trace.currentPath()).toBe(oldPath);
});
it("defers trace rotation after boundary compaction until the next run writes", async () => {
const llm1 = new ScriptedLLM(
[
{ messages: [assistantText("answer"), usage(150, 150)] },
{ messages: [assistantText("<summary>s</summary>")] },
],
"llm1",
);
const llm2 = new ScriptedLLM(
[{ messages: [assistantText("next done"), usage(10, 160)] }],
"llm2",
);
const trace = new Writer({ tracesDir: traces, sessionId: "sess_lazy" });
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
trace,
sessionMeta: metaMessage,
compaction: settings(),
createLLM: () => llm2,
});
const oldPath = trace.currentPath();
await collect(engine.run([userText("task 1")], { approve: allowAll }));
// A new file is not created right after boundary compaction completes: the current path is unchanged, and only the old file exists on disk.
expect(trace.currentPath()).toBe(oldPath);
expect(await readdir(dirname(oldPath))).toEqual(["sess_lazy_001.jsonl"]);
await collect(engine.run([userText("task 2")], { approve: allowAll }));
// Rotation happens only once the next round has a message to write: the new file opens with session_meta, followed by the summary and the new prompt.
expect(trace.currentPath()).not.toBe(oldPath);
const newTrace = await readTrace(trace.currentPath());
expect(newTrace[0]!.type).toBe("session_meta");
expect(
((newTrace[1]!.payload as { text?: string }).text ?? "").startsWith("<context_summary>"),
).toBe(true);
expect((newTrace[2]!.payload as { text?: string }).text).toBe("task 2");
});
it("reconnect exhaustion on the compaction request converges to failed", async () => {
const llm1 = new ScriptedLLM(
[
{ messages: [assistantText("answer"), usage(150, 150)] },
{ messages: [], outcome: { status: "timeout" } },
{ messages: [], outcome: { status: "timeout" } },
],
"llm1",
);
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
createLLM: () => llm1,
maxReconnects: 1,
reconnectBackoffMs: 1,
});
const out = await collect(engine.run([userText("go")], { approve: allowAll }));
const events = compactionEvents(out);
expect(events[1]).toMatchObject({ type: "compaction_end", status: "failed" });
// The retry resends the original input (tool results + prompt; here there are no tool results, just the prompt).
expect(llm1.calls).toHaveLength(3);
expect(payloadTypes(llm1.calls[2]!)).toEqual(["text"]);
});
it("session turns reaching (==) the threshold compact at task end — no waiting for the next task", async () => {
const llm1 = new ScriptedLLM(
[
// Task 1: two LLM requests (a tool round + the final reply).
// After round 1, turns=1 < 2 doesn't trigger; round 2 (task wrap-up), turns=2 >= 2 -> compacts immediately.
{
messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(10, 10)],
},
{ messages: [assistantText("t1 done"), usage(10, 20)] },
// Compaction request (sent out immediately when task 1 ends).
{ messages: [assistantText("<summary>s</summary>")] },
],
"llm1",
);
const llm2 = new ScriptedLLM([{ messages: [assistantText("t2 done"), usage(10, 30)] }], "llm2");
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings({ maxContextLength: -1, maxSessionTurns: 2 }),
createLLM: () => llm2,
});
// Triggers right at task 1's wrap-up (compacts as soon as the threshold is reached, without waiting for the next task); the summary request goes to the old instance.
const out1 = await collect(engine.run([userText("task 1")], { approve: allowAll }));
const events = compactionEvents(out1);
expect(events[0]).toMatchObject({ type: "compaction_begin", reason: "turns", turns: 2 });
expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" });
expect(llm1.calls).toHaveLength(3);
// The counter resets after compaction completes: task 2 is picked up by the new instance (summary + new prompt), and no further trigger fires.
const out2 = await collect(engine.run([userText("task 2")], { approve: allowAll }));
expect(compactionEvents(out2)).toHaveLength(0);
expect(llm2.calls).toHaveLength(1);
expect(llm2.calls[0]!.map(textOf)).toEqual([
"<context_summary>\ns\n</context_summary>",
"task 2",
]);
});
it("context usage exactly equal to the threshold triggers compaction (>=, not >)", async () => {
const llm1 = new ScriptedLLM(
[
// Wrap-up round context usage 100 == threshold 100 -> triggers.
{ messages: [assistantText("answer"), usage(100, 100)] },
{ messages: [assistantText("<summary>eq</summary>")] },
],
"llm1",
);
const llm2 = new ScriptedLLM([], "llm2");
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
createLLM: () => llm2,
});
const out = await collect(engine.run([userText("go")], { approve: allowAll }));
const events = compactionEvents(out);
expect(events[0]).toMatchObject({
type: "compaction_begin",
reason: "context",
context: 100,
});
expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" });
});
it("discard defers mid-task, then swaps the LLM at task end without a compaction request", async () => {
const llm1 = new ScriptedLLM(
[
// Round 1: over the limit, but the task is still in progress -> deferred.
{
messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)],
},
// Round 2: task ends -> performs discard (no compaction request sent).
{ messages: [assistantText("done"), usage(160, 310)] },
],
"llm1",
);
const llm2 = new ScriptedLLM([{ messages: [assistantText("fresh"), usage(10, 320)] }], "llm2");
const trace = new Writer({ tracesDir: traces, sessionId: "sess_discard" });
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
trace,
sessionMeta: metaMessage,
compaction: settings({ mode: "discard" }),
createLLM: () => llm2,
});
const oldPath = trace.currentPath();
const out = await collect(engine.run([userText("go")], { approve: allowAll }));
// Triggers exactly once, at task end; the old LLM is called exactly twice (no compaction request).
const events = compactionEvents(out);
expect(events.map((e) => `${e.type}:${e.mode}`)).toEqual([
"compaction_begin:discard",
"compaction_end:discard",
]);
expect(llm1.calls).toHaveLength(2);
// The next round's input is used as-is as the new instance's first input (no <context_summary>).
await collect(engine.run([userText("next task")], { approve: allowAll }));
expect(llm2.calls[0]!.map(textOf)).toEqual(["next task"]);
// Trace splits into files: the new file starts with session_meta.
const newTrace = await readTrace(trace.currentPath());
expect(trace.currentPath()).not.toBe(oldPath);
expect(newTrace[0]!.type).toBe("session_meta");
});
it("lenient extraction: output without <summary> tags is used verbatim", async () => {
const llm1 = new ScriptedLLM(
[
{ messages: [assistantText("answer"), usage(150, 150)] },
{ messages: [assistantText("plain summary text without tags")] },
],
"llm1",
);
const llm2 = new ScriptedLLM([{ messages: [assistantText("ok"), usage(10, 160)] }], "llm2");
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
createLLM: () => llm2,
});
await collect(engine.run([userText("go")], { approve: allowAll }));
await collect(engine.run([userText("next")], { approve: allowAll }));
expect(textOf(llm2.calls[0]![0]!)).toBe(
"<context_summary>\nplain summary text without tags\n</context_summary>",
);
});
it("manual compaction skips threshold checks and reuses the same flow", async () => {
const llm1 = new ScriptedLLM(
[
// One ordinary task (well under the limit).
{ messages: [assistantText("small"), usage(10, 10)] },
// Manual compaction request.
{ messages: [assistantText("<summary>manual s</summary>")] },
],
"llm1",
);
const llm2 = new ScriptedLLM([{ messages: [assistantText("after"), usage(5, 20)] }], "llm2");
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
createLLM: () => llm2,
});
await collect(engine.run([userText("hi")], { approve: allowAll }));
const out = await collect(engine.compact());
const events = compactionEvents(out);
expect(events[0]).toMatchObject({ type: "compaction_begin", reason: "manual" });
expect(events[1]).toMatchObject({ type: "compaction_end", status: "completed" });
await collect(engine.run([userText("next")], { approve: allowAll }));
expect(llm2.calls[0]!.map(textOf)).toEqual([
"<context_summary>\nmanual s\n</context_summary>",
"next",
]);
});
it("no compaction capability (createLLM missing) means thresholds never fire and compact() is a no-op", async () => {
const llm1 = new ScriptedLLM(
[{ messages: [assistantText("big"), usage(999999, 999999)] }],
"llm1",
);
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
});
const out = await collect(engine.run([userText("go")], { approve: allowAll }));
expect(compactionEvents(out)).toHaveLength(0);
expect(await collect(engine.compact())).toHaveLength(0);
expect(llm1.calls).toHaveLength(1);
});
it("compactability(): 逐一给出「不能压缩」的原因,而不是笼统的 false", async () => {
// compact() emits **zero messages** when there is nothing compactable. The caller (Web / CLI)
// must be able to ask ahead of time, otherwise it can only wait forever for a compaction
// banner that never comes -- exactly how "no response from /compact after interrupting on the
// web" happens: interrupt the first request -> token_usage is never received -> sessionTurns
// stays at 0 -> compact() returns immediately.
// The reason also needs to distinguish "just compacted" from "haven't chatted yet" -- both
// have sessionTurns == 0, but they're two completely different messages to the user: telling
// someone who just finished compacting that there's "no completed conversation turn yet" is
// effectively saying nothing useful.
// The compaction request goes to the **current** LLM (the script's second entry); createLLM
// supplies the LLM used for the new context after compaction.
const llm1 = new ScriptedLLM(
[
// Usage is kept under maxContextLength (100 per settings()) so automatic compaction doesn't jump in first and reset sessionTurns.
{ messages: [assistantText("hi"), usage(10, 10)] },
{ messages: [assistantText("<summary>s</summary>")] }, // manual compaction request
],
"llm1",
);
const llm2 = new ScriptedLLM([{ messages: [assistantText("after"), usage(5, 20)] }], "llm2");
// (1) No compaction capability configured (no compaction / createLLM).
const noCap = new ContextEngine({ llm: llm1, environment: fakeEnvironment });
expect(noCap.compactability()).toBe("unsupported");
// (2) Capability configured, but the current context hasn't finished a single round yet: not compactable, and compact() indeed emits no messages.
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
createLLM: () => llm2,
});
expect(engine.compactability()).toBe("empty");
expect(await collect(engine.compact())).toHaveLength(0);
// (3) Compactable only after a round finishes (token_usage received).
await collect(engine.run([userText("go")], { approve: allowAll }));
expect(engine.compactability()).toBe("ok");
expect(compactionEvents(await collect(engine.compact()))).not.toHaveLength(0);
// (4) Compacting again right after a compaction: also not compactable (the new context is
// empty), but the reason is "just compacted" -- it must not say "no completed conversation
// turn yet" again, since the user clearly just finished a whole round.
expect(engine.compactability()).toBe("just_compacted");
expect(await collect(engine.compact())).toHaveLength(0);
// (5) Compactable again once a round finishes in the new context.
await collect(engine.run([userText("next")], { approve: allowAll }));
expect(engine.compactability()).toBe("ok");
});
it("user abort during the compaction request keeps the context and carries tool outputs over", async () => {
const controller = new AbortController();
const llm1 = new ScriptedLLM(
[
{
messages: [toolCall({ name: "t", arguments: "{}", toolCallId: "c1" }), usage(150, 150)],
},
{ messages: [], outcome: { status: "aborted" } },
],
"llm1",
);
const engine = new ContextEngine({
llm: llm1,
environment: fakeEnvironment,
compaction: settings(),
createLLM: () => llm1,
});
// The abort signal is already pending before the compaction request: the fake LLM finishes straight to aborted.
const approveThenAbort: ApproveFn = async () => "allow";
const runGen = engine.run([userText("go")], {
approve: approveThenAbort,
signal: controller.signal,
});
const out: OmniMessage[] = [];
for await (const msg of runGen) {
out.push(msg);
// Simulates a user abort right after the compaction start event (the compaction request then returns aborted).
const p = msg.payload as { type?: string };
if (p.type === "compaction_begin") controller.abort();
}
const events = compactionEvents(out);
expect(events[1]).toMatchObject({ type: "compaction_end", status: "aborted" });
// Interrupt cleanup: the tool output is held as carry-over per case A, and the run wraps up with an abort event.
expect(payloadTypes(out)).toContain("abort");
});
});
+241
View File
@@ -0,0 +1,241 @@
/**
* Unit tests (offline) for the read_image "vision-model describe" variant, driven by a fake LLM:
* definition overrides (new prompt parameter, description mentioning the vision model id), a
* single image + prompt sent to the vision model, text output and failure paths (no vision model
* configured / vision model request fails), results carrying no images; and swapping on the
* Environment side (injecting visionDescriber switches to the describe variant).
*/
import { afterEach, beforeEach, describe, expect, it } from "vitest";
import { mkdtemp, rm, writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import path from "node:path";
import {
DESCRIBE_IMAGE_NAME,
createDescribeImageTool,
} from "../src/environment/tools/describe-image.js";
import { BUILTIN_TOOL_FACTORIES } from "../src/environment/tools/registry.js";
import { Environment } from "../src/environment/environment.js";
import { assistantText, partialText, toolCall } from "../src/omnimessage/index.js";
import type { OmniMessage } from "../src/omnimessage/index.js";
import type { ToolResult } from "../src/environment/tools/types.js";
import type {
GenerativeModelParameters,
LLMInterface,
LLMOutcome,
ToolDefinitionConfig,
VisionDescriberService,
} from "../src/interfaces.js";
/** 1x1 transparent PNG (includes magic bytes, enough for mime sniffing). */
const PNG_1X1 = Buffer.from(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==",
"base64",
);
/** Config entry for describe_image (forModel: "text-only"; the definition comes entirely from config, the implementation never rewrites it at runtime). */
const definition: ToolDefinitionConfig = {
name: DESCRIBE_IMAGE_NAME,
forModel: "text-only",
description: "describe image via vision model",
parameters: {
type: "object",
properties: { source: { type: "string" }, prompt: { type: "string" } },
required: ["source"],
},
permission: "r",
};
/**
* Fake vision LLM: records the received newMessages, emits output following the real streaming
* protocol (partial start -> word-by-word delta -> stop -> complete text), and finishes with the given outcome.
*/
function fakeLLM(reply: string, outcome: LLMOutcome = { status: "completed" }) {
const calls: GenerativeModelParameters[] = [];
const llm: LLMInterface = {
// eslint-disable-next-line @typescript-eslint/require-await
async *streamGenerate(params: GenerativeModelParameters) {
calls.push(params);
if (reply) {
yield partialText("start");
// Split into two delta chunks to verify piecewise forwarding (rather than buffering the whole thing).
const mid = Math.ceil(reply.length / 2);
yield partialText("delta", reply.slice(0, mid));
yield partialText("delta", reply.slice(mid));
yield partialText("stop");
yield assistantText(reply);
}
return outcome;
},
};
return { llm, calls };
}
async function run(
args: Record<string, unknown>,
workspaceDir: string,
describer: VisionDescriberService,
) {
const tool = createDescribeImageTool(definition, describer);
const gen = tool.execute(args, { workspaceDir, toolCallId: "c1" });
const messages: OmniMessage[] = [];
let result: ToolResult | void;
for (;;) {
const res = await gen.next();
if (res.done) {
result = res.value;
break;
}
messages.push(res.value);
}
const text = messages.map((m) => (m.payload as { output?: string }).output ?? "").join("");
return { messages, result, text };
}
let tmp: string;
beforeEach(async () => {
tmp = await mkdtemp(path.join(tmpdir(), "penguin-descimg-"));
});
afterEach(async () => {
await rm(tmp, { recursive: true, force: true });
});
describe("describe_image(read_image 的纯文本模型版)", () => {
it("定义原样取自配置条目(不做运行期改写)", () => {
const tool = createDescribeImageTool(definition, { modelId: "vis-1" });
expect(tool.name).toBe(DESCRIBE_IMAGE_NAME);
expect(tool.definition).toBe(definition);
});
it("registry 按工具名装配;未注入 visionDescriber 时以 failed 说明收尾(而非回图片)", async () => {
const factory = BUILTIN_TOOL_FACTORIES[DESCRIBE_IMAGE_NAME]!;
await writeFile(path.join(tmp, "a.png"), PNG_1X1);
const describeTool = factory(definition, undefined);
const gen = describeTool.execute({ source: "a.png" }, { workspaceDir: tmp, toolCallId: "c1" });
let result: ToolResult | void;
let text = "";
for (;;) {
const res = await gen.next();
if (res.done) {
result = res.value;
break;
}
text += (res.value.payload as { output?: string }).output ?? "";
}
expect(result?.stopReason).toBe("failed");
expect(text).toContain("No vision model");
});
it("图片 + 自定义 prompt 单发视觉模型,回其文本,结果不携带 images", async () => {
await writeFile(path.join(tmp, "a.png"), PNG_1X1);
const { llm, calls } = fakeLLM("图里是一只企鹅。");
const describer: VisionDescriberService = { modelId: "vis-1", createLLM: () => llm };
const { messages, result, text } = await run(
{ source: "a.png", prompt: "图里是什么动物?" },
tmp,
describer,
);
// Single message = prompt text + data URL image (same role user, merged into one request).
expect(calls).toHaveLength(1);
const payloads = calls[0]!.newMessages.map(
(m) => m.payload as { type: string; text?: string; image_url?: string },
);
expect(payloads[0]!.type).toBe("text");
expect(payloads[0]!.text).toBe("图里是什么动物?");
expect(payloads[1]!.type).toBe("image_url");
expect(payloads[1]!.image_url).toBe(`data:image/png;base64,${PNG_1X1.toString("base64")}`);
expect(text).toContain("described by vis-1");
expect(text).toContain("图里是一只企鹅。");
// Streaming forward: the header line and description deltas are emitted as separate chunks (not buffered as a whole), and the complete text is not forwarded again.
const outputs = messages.map((m) => (m.payload as { output?: string }).output ?? "");
expect(outputs.length).toBeGreaterThanOrEqual(3); // header + >=2 description delta chunks
expect(outputs[0]).toContain("described by vis-1");
expect(outputs.slice(1).join("")).toBe("图里是一只企鹅。");
// Text-based description: the result carries no images (images never enter session history).
expect(result?.images).toBeUndefined();
expect(result?.stopReason).toBeUndefined(); // defaults to completed
});
it("未给 prompt 时用默认问题", async () => {
await writeFile(path.join(tmp, "a.png"), PNG_1X1);
const { llm, calls } = fakeLLM("desc");
await run({ source: "a.png" }, tmp, { modelId: "vis-1", createLLM: () => llm });
const first = calls[0]!.newMessages[0]!.payload as { text?: string };
expect(first.text).toContain("Describe this image");
});
it("未配置视觉模型:failed 并解释如何配置", async () => {
await writeFile(path.join(tmp, "a.png"), PNG_1X1);
const { result, text } = await run({ source: "a.png" }, tmp, { modelId: null });
expect(result?.stopReason).toBe("failed");
expect(text).toContain("No vision model");
expect(text).toContain("vision_model");
});
it("视觉模型请求失败:failed 并带状态与消息", async () => {
await writeFile(path.join(tmp, "a.png"), PNG_1X1);
const { llm } = fakeLLM("", { status: "failed", message: "401 unauthorized" });
const { result, text } = await run({ source: "a.png" }, tmp, {
modelId: "vis-1",
createLLM: () => llm,
});
expect(result?.stopReason).toBe("failed");
expect(text).toContain("failed");
expect(text).toContain("401 unauthorized");
});
it("图片校验复用 read_image:不支持的类型直接 failed,不请求视觉模型", async () => {
await writeFile(path.join(tmp, "a.txt"), "not an image");
const { llm, calls } = fakeLLM("desc");
const { result } = await run({ source: "a.txt" }, tmp, {
modelId: "vis-1",
createLLM: () => llm,
});
expect(result?.stopReason).toBe("failed");
expect(calls).toHaveLength(0);
});
it("Environment 装配 describe_image 条目走代读实现,定义与配置一致", async () => {
await writeFile(path.join(tmp, "a.png"), PNG_1X1);
const { llm } = fakeLLM("代读结果");
const env = new Environment({
workspaceDir: tmp,
toolConfig: {
// Already filtered by selectBuiltinToolsForModel per session model before assembly; only the describe entry remains here.
customTools: [definition],
mcpServers: [],
},
services: { visionDescriber: { modelId: "vis-1", createLLM: () => llm } },
});
// Tool listing matches config: describe_image carries the prompt parameter.
const tools = await env.listTools();
const describeImage = tools.find((t) => t.name === DESCRIBE_IMAGE_NAME)!;
const props = (describeImage.parameters as { properties: Record<string, unknown> }).properties;
expect(Object.keys(props)).toContain("prompt");
// Execution goes through description: outputs text, the complete message carries no images.
const out: OmniMessage[] = [];
for await (const m of env.executeTool({
toolCall: toolCall({
name: DESCRIBE_IMAGE_NAME,
arguments: '{"source":"a.png"}',
toolCallId: "t1",
}),
})) {
out.push(m);
}
const complete = out[out.length - 1]!.payload as {
type?: string;
output?: string;
images?: string[];
stop_reason?: string;
};
expect(complete.type).toBe("tool_call_output");
expect(complete.stop_reason).toBe("completed");
expect(complete.output).toContain("代读结果");
expect(complete.images).toBeUndefined();
});
});
File diff suppressed because it is too large Load Diff

Some files were not shown because too many files have changed in this diff Show More